返回 DeepSeek-Reasonix
encoding_helpers.go
根目录 / internal / tool / builtin / encoding_helpers.go
1 package builtin
2
3 import (
4 "fmt"
5 "os"
6 "slices"
7 "strings"
8
9 "reasonix/internal/fileutil"
10 fileenc "reasonix/internal/fileutil/encoding"
11 "reasonix/internal/tool"
12 )
13
14 // readFileEncoded reads a file and decodes its encoding to UTF-8.
15 // Returns the decoded content and the detected encoding kind so callers
16 // can re-encode on write to preserve the original charset.
17 func readFileEncoded(path string) (content string, enc fileenc.Kind, err error) {
18 b, err := os.ReadFile(path)
19 if err != nil {
20 return "", 0, err
21 }
22 enc, text := fileenc.DetectAndDecode(b)
23 return string(text), enc, nil
24 }
25
26 // writeFileEncoded encodes content back to the given encoding and writes it.
27 // The write is atomic: a truncating write that fails midway (a Windows filter
28 // driver holding a transient lock, a full disk) would leave the user's source
29 // file empty or half-written.
30 func writeFileEncoded(path string, content string, enc fileenc.Kind) error {
31 data, err := fileenc.Encode(content, enc)
32 if err != nil {
33 return err
34 }
35 return fileutil.AtomicOverwriteFileStrict(path, data, 0o644)
36 }
37
38 // matchLineEndings adapts an edit's old/new text to a CRLF file when the literal
39 // old_string isn't present but its CRLF form is. read_file strips '\r' (bufio
40 // ScanLines), so a model's multi-line old_string arrives LF-only while a
41 // Windows/CJK source stores '\r\n'; rewriting search and replacement to the
42 // file's ending fixes the match without rewriting the file's other line endings.
43 func matchLineEndings(content, old, new string) (string, string) {
44 if strings.Contains(content, old) || !strings.Contains(content, "\r\n") {
45 return old, new
46 }
47 if strings.Contains(content, toCRLF(old)) {
48 return toCRLF(old), toCRLF(new)
49 }
50 return old, new
51 }
52
53 func toCRLF(s string) string {
54 return strings.ReplaceAll(strings.ReplaceAll(s, "\r\n", "\n"), "\n", "\r\n")
55 }
56
57 func matchReplacementLineEndings(content, replacement string) string {
58 if strings.Contains(content, "\r\n") {
59 return toCRLF(replacement)
60 }
61 return replacement
62 }
63
64 type editApplyResult struct {
65 updated string
66 applied int
67 matches int
68 fuzzy bool
69 receipt editReplacementReceipt
70 }
71
72 type editRange struct {
73 start int
74 end int
75 }
76
77 // editReplacementReceipt records only the span the tool actually matched and
78 // the span it wrote in its place. It deliberately excludes surrounding file
79 // content so a successful edit can ground the next model turn without widening
80 // provider-visible workspace data.
81 type editReplacementReceipt struct {
82 matched string
83 replacement string
84 occurrences int
85 fuzzy bool
86 }
87
88 // applyOldStringEdit is the shared edit_file/multi_edit/Preview contract. It
89 // preserves the exact-match rule first, then falls back to a narrow fuzzy match
90 // for the mismatches read_file commonly introduces or hides: trailing
91 // whitespace, tab-vs-spaces indentation, and copied read_file line prefixes.
92 // Non-replace_all edits still require exactly one match, including fuzzy
93 // matches.
94 func applyOldStringEdit(content, oldString, newString string, replaceAll bool) editApplyResult {
95 old, newStr := matchLineEndings(content, oldString, newString)
96 if replaceAll {
97 if count := strings.Count(content, old); count > 0 {
98 return editApplyResult{
99 updated: strings.ReplaceAll(content, old, newStr),
100 applied: count,
101 matches: count,
102 receipt: editReplacementReceipt{
103 matched: old,
104 replacement: newStr,
105 occurrences: count,
106 },
107 }
108 }
109 ranges := fuzzyEditRanges(content, old)
110 if len(ranges) == 0 {
111 return editApplyResult{updated: content}
112 }
113 replacement := matchReplacementLineEndings(content, newStr)
114 return editApplyResult{
115 updated: replaceEditRanges(content, ranges, replacement),
116 applied: len(ranges),
117 matches: len(ranges),
118 fuzzy: true,
119 receipt: editReplacementReceipt{
120 matched: matchedRangeSample(content, old, ranges),
121 replacement: replacement,
122 occurrences: len(ranges),
123 fuzzy: true,
124 },
125 }
126 }
127
128 switch count := strings.Count(content, old); count {
129 case 0:
130 ranges := fuzzyEditRanges(content, old)
131 if len(ranges) != 1 {
132 return editApplyResult{updated: content, matches: len(ranges)}
133 }
134 return editApplyResult{
135 updated: replaceEditRanges(content, ranges, matchReplacementLineEndings(content, newStr)),
136 applied: 1,
137 matches: 1,
138 fuzzy: true,
139 receipt: editReplacementReceipt{
140 matched: matchedRangeSample(content, old, ranges),
141 replacement: matchReplacementLineEndings(content, newStr),
142 occurrences: 1,
143 fuzzy: true,
144 },
145 }
146 case 1:
147 return editApplyResult{
148 updated: strings.Replace(content, old, newStr, 1),
149 applied: 1,
150 matches: 1,
151 receipt: editReplacementReceipt{
152 matched: old,
153 replacement: newStr,
154 occurrences: 1,
155 },
156 }
157 default:
158 return editApplyResult{updated: content, matches: count}
159 }
160 }
161
162 func matchedRangeSample(content, fallback string, ranges []editRange) string {
163 if len(ranges) == 0 {
164 return fallback
165 }
166 r := ranges[0]
167 if r.start < 0 || r.end < r.start || r.end > len(content) {
168 return fallback
169 }
170 actual := content[r.start:r.end]
171 sample := clipPostWriteSpan(actual, maxCapturedReceiptSpanBytes)
172 if len(sample) == len(actual) {
173 // Do not let a short substring keep an otherwise-dead large intermediate
174 // multi_edit buffer alive until all later steps finish.
175 return strings.Clone(sample)
176 }
177 return sample
178 }
179
180 func oldStringNotFoundError(path, oldString, content string) (err error) {
181 defer func() {
182 err = &tool.OperationError{Diagnostic: tool.OperationDiagnostic{Code: tool.WriteEvidenceStale, Path: path, Recovery: "re-read the target range, then retry with its current text"}, Cause: err}
183 }()
184 hint := oldStringNotFoundHint(oldString, content)
185 if line, text, ok := nearestContentLine(oldString, content); ok {
186 return fmt.Errorf("old_string not found in %s (nearest line %d: %q).%s", path, line, text, hint)
187 }
188 return fmt.Errorf("old_string not found in %s.%s", path, hint)
189 }
190
191 func oldStringNotFoundHint(oldString, content string) string {
192 base := " Re-read the current file before retrying; if several related edits target the same area, combine the final replacements in one multi_edit call."
193 if !strings.Contains(content, "\r\n") {
194 return base
195 }
196 normalizedContent := strings.ReplaceAll(content, "\r\n", "\n")
197 normalizedOld := strings.ReplaceAll(oldString, "\r\n", "\n")
198 if strings.Contains(normalizedContent, normalizedOld) {
199 return " The target file uses CRLF line endings; edit_file/multi_edit normally normalize LF-only old_string for CRLF files, so this is likely stale context. Re-read the current file before retrying."
200 }
201 return " The target file uses CRLF line endings, but edit_file/multi_edit already tolerate LF-only old_string for CRLF files; check for stale, incomplete, or non-unique context before retrying."
202 }
203
204 func oldStringNotUniqueError(path, oldString, content string, matches int, replaceAllHint bool) (err error) {
205 defer func() {
206 err = &tool.OperationError{Diagnostic: tool.OperationDiagnostic{Code: tool.WriteTargetAmbiguous, Path: path, Recovery: "read surrounding lines and use a unique anchor"}, Cause: err}
207 }()
208 lineHint := oldStringMatchLineSummary(oldString, content, 5)
209 if replaceAllHint {
210 return fmt.Errorf("old_string is not unique in %s (%d matches)%s; add nearby unique code, not just repeated separator lines, or set replace_all if every match should change", path, matches, lineHint)
211 }
212 return fmt.Errorf("old_string is not unique in %s (%d matches)%s; add nearby unique code, not just repeated separator lines", path, matches, lineHint)
213 }
214
215 type lineSegment struct {
216 raw string
217 start int
218 end int
219 }
220
221 type fuzzyMode struct {
222 stripOldReadPrefixes bool
223 trimTrailing bool
224 expandTabs bool
225 trimLeading bool
226 }
227
228 func fuzzyEditRanges(content, old string) []editRange {
229 if old == "" || content == "" {
230 return nil
231 }
232 contentLines := splitLineSegments(content)
233 oldLines := splitLineSegments(old)
234 if len(oldLines) == 0 || len(oldLines) > len(contentLines) {
235 return nil
236 }
237
238 oldHasReadPrefixes := allLinesHaveReadFilePrefix(oldLines)
239 modes := []fuzzyMode{
240 {trimTrailing: true},
241 {trimTrailing: true, expandTabs: true},
242 }
243 if oldHasReadPrefixes {
244 modes = append(modes,
245 fuzzyMode{stripOldReadPrefixes: true, trimTrailing: true},
246 fuzzyMode{stripOldReadPrefixes: true, trimTrailing: true, expandTabs: true},
247 )
248 }
249
250 for _, mode := range modes {
251 normOld := make([]string, len(oldLines))
252 for i, line := range oldLines {
253 normOld[i] = normalizeFuzzyLine(line.raw, lineHasNewline(line.raw), mode, mode.stripOldReadPrefixes)
254 }
255 var ranges []editRange
256 for i := 0; i <= len(contentLines)-len(oldLines); {
257 if fuzzyWindowMatches(contentLines[i:i+len(oldLines)], oldLines, normOld, mode) {
258 ranges = append(ranges, editRange{
259 start: contentLines[i].start,
260 end: fuzzyWindowEnd(contentLines[i+len(oldLines)-1], oldLines[len(oldLines)-1]),
261 })
262 i += len(oldLines)
263 continue
264 }
265 i++
266 }
267 if len(ranges) > 0 {
268 return ranges
269 }
270 }
271 return nil
272 }
273
274 func fuzzyWindowMatches(contentWindow, oldLines []lineSegment, normOld []string, mode fuzzyMode) bool {
275 for i, contentLine := range contentWindow {
276 oldHasNewline := lineHasNewline(oldLines[i].raw)
277 if oldHasNewline && !lineHasNewline(contentLine.raw) {
278 return false
279 }
280 got := normalizeFuzzyLine(contentLine.raw, oldHasNewline, mode, false)
281 if got != normOld[i] {
282 return false
283 }
284 }
285 return true
286 }
287
288 func splitLineSegments(s string) []lineSegment {
289 if s == "" {
290 return nil
291 }
292 var lines []lineSegment
293 start := 0
294 for i, r := range s {
295 if r == '\n' {
296 end := i + 1
297 lines = append(lines, lineSegment{raw: s[start:end], start: start, end: end})
298 start = end
299 }
300 }
301 if start < len(s) {
302 lines = append(lines, lineSegment{raw: s[start:], start: start, end: len(s)})
303 }
304 return lines
305 }
306
307 func lineHasNewline(line string) bool {
308 return strings.HasSuffix(line, "\n")
309 }
310
311 func fuzzyWindowEnd(contentLast, oldLast lineSegment) int {
312 if lineHasNewline(oldLast.raw) || !lineHasNewline(contentLast.raw) {
313 return contentLast.end
314 }
315 end := contentLast.end - 1
316 if end > contentLast.start && contentLast.raw[len(contentLast.raw)-2] == '\r' {
317 end--
318 }
319 return end
320 }
321
322 func normalizeFuzzyLine(line string, includeNewline bool, mode fuzzyMode, stripReadPrefix bool) string {
323 body := strings.TrimSuffix(line, "\n")
324 if stripReadPrefix {
325 body, _ = stripReadFileLinePrefix(body)
326 }
327 if mode.trimTrailing {
328 body = strings.TrimRight(body, " \t\r")
329 }
330 if mode.expandTabs {
331 body = strings.ReplaceAll(body, "\t", " ")
332 }
333 if mode.trimLeading {
334 body = strings.TrimLeft(body, " \t")
335 }
336 if includeNewline {
337 return body + "\n"
338 }
339 return body
340 }
341
342 func allLinesHaveReadFilePrefix(lines []lineSegment) bool {
343 if len(lines) == 0 {
344 return false
345 }
346 for _, line := range lines {
347 body := strings.TrimSuffix(line.raw, "\n")
348 if _, ok := stripReadFileLinePrefix(body); !ok {
349 return false
350 }
351 }
352 return true
353 }
354
355 func stripReadFileLinePrefix(line string) (string, bool) {
356 i := 0
357 for i < len(line) && (line[i] == ' ' || line[i] == '\t') {
358 i++
359 }
360 j := i
361 for j < len(line) && line[j] >= '0' && line[j] <= '9' {
362 j++
363 }
364 if j == i || !strings.HasPrefix(line[j:], "\u2192") {
365 return line, false
366 }
367 return line[j+len("\u2192"):], true
368 }
369
370 func replaceEditRanges(content string, ranges []editRange, replacement string) string {
371 updated := content
372 for _, v := range slices.Backward(ranges) {
373 r := v
374 updated = updated[:r.start] + replacement + updated[r.end:]
375 }
376 return updated
377 }
378
379 func nearestContentLine(oldString, content string) (int, string, bool) {
380 oldLines := splitLineSegments(oldString)
381 if len(oldLines) == 0 {
382 return 0, "", false
383 }
384 target := strings.TrimSpace(normalizeFuzzyLine(oldLines[0].raw, false, fuzzyMode{trimTrailing: true, expandTabs: true}, true))
385 if target == "" {
386 return 0, "", false
387 }
388 bestLine := 0
389 bestScore := 0
390 bestText := ""
391 for i, line := range splitLineSegments(content) {
392 text := strings.TrimSuffix(line.raw, "\n")
393 score := commonPrefixLen(strings.TrimSpace(strings.ReplaceAll(text, "\t", " ")), target)
394 if score > bestScore {
395 bestLine = i + 1
396 bestScore = score
397 bestText = text
398 }
399 }
400 if bestScore < 3 {
401 return 0, "", false
402 }
403 return bestLine, bestText, true
404 }
405
406 func oldStringMatchLineSummary(oldString, content string, limit int) string {
407 if limit <= 0 {
408 return ""
409 }
410 target := firstNonEmptyLine(oldString)
411 if target == "" {
412 return ""
413 }
414 var matches []int
415 for i, line := range splitLineSegments(content) {
416 text := strings.TrimSuffix(line.raw, "\n")
417 text = strings.TrimSuffix(text, "\r")
418 if strings.Contains(text, target) {
419 matches = append(matches, i+1)
420 }
421 }
422 if len(matches) == 0 {
423 return ""
424 }
425 var b strings.Builder
426 b.WriteString("; matching lines include ")
427 for i, line := range matches {
428 if i >= limit {
429 b.WriteString(", ...")
430 break
431 }
432 if i > 0 {
433 b.WriteString(", ")
434 }
435 fmt.Fprint(&b, line)
436 }
437 return b.String()
438 }
439
440 func firstNonEmptyLine(s string) string {
441 for _, line := range splitLineSegments(s) {
442 text := strings.TrimSpace(strings.TrimSuffix(line.raw, "\n"))
443 text = strings.TrimSuffix(text, "\r")
444 if text != "" {
445 return text
446 }
447 }
448 return ""
449 }
450
451 func commonPrefixLen(a, b string) int {
452 n := min(len(b), len(a))
453 for i := range n {
454 if a[i] != b[i] {
455 return i
456 }
457 }
458 return n
459 }
460
460 lines GO