返回 DeepSeek-Reasonix
encoding.go
根目录 / internal / fileutil / encoding / encoding.go
1 // Package encoding detects and converts file encodings for the built-in
2 // file tools. The detection cascade (BOM → strict UTF-8 → GB18030 → GBK →
3 // lossy UTF-8) mirrors v1's file-encoding.ts and keeps CJK Windows files editable
4 // without silently mangling their bytes.
5 package encoding
6
7 import (
8 "bytes"
9 "encoding/binary"
10 "os"
11
12 "golang.org/x/text/encoding/unicode"
13 "golang.org/x/text/transform"
14 )
15
16 // Kind identifies a detected file encoding.
17 type Kind int
18
19 const (
20 // UTF8 is plain UTF-8 without a BOM — the common case.
21 UTF8 Kind = iota
22 // UTF8BOM is UTF-8 with a leading BOM (EF BB BF).
23 UTF8BOM
24 // UTF16LE is UTF-16 Little-Endian with a BOM (FF FE).
25 UTF16LE
26 // UTF16BE is UTF-16 Big-Endian with a BOM (FE FF).
27 UTF16BE
28 // GB18030 is the Chinese national standard charset (superset of GBK).
29 GB18030
30 // LossyUTF8 is not valid UTF-8 and no charset restores it byte for byte.
31 // Its bytes pass through untouched, so a rewrite keeps what it did not edit.
32 LossyUTF8
33 // UTF16LENoBOM is UTF-16 Little-Endian without a BOM — common for source
34 // files saved by Windows tools. Detected heuristically from the NUL-byte
35 // pattern; written back without a BOM to preserve the original bytes.
36 UTF16LENoBOM
37 // UTF16BENoBOM is UTF-16 Big-Endian without a BOM.
38 UTF16BENoBOM
39 // GBK is CP936 where GB18030 cannot restore the bytes, such as its 0x80 euro.
40 // Appended: checkpoints persist a Kind by value.
41 GBK
42 )
43
44 var utf8BOM = []byte{0xEF, 0xBB, 0xBF}
45
46 // Detect returns the encoding kind for the given raw file bytes. The same
47 // bytes should then be passed to Decode for conversion to UTF-8.
48 func Detect(data []byte) (Kind, []byte) {
49 k, _, _ := sniff(data, true)
50 return k, data
51 }
52
53 // DetectQuick checks only for BOM prefixes in the first few bytes. This is
54 // the fast path for peek-based binary rejection: BOM-prefixed files (UTF-16,
55 // UTF-8 BOM) skip the NUL-byte check since 0x00 is normal in UTF-16. Returns
56 // UTF8 for non-BOM content (the caller should fall through to full Detect
57 // after verifying no NUL bytes).
58 func DetectQuick(peek []byte) Kind {
59 switch {
60 case len(peek) >= 3 && peek[0] == 0xEF && peek[1] == 0xBB && peek[2] == 0xBF:
61 return UTF8BOM
62 case len(peek) >= 2 && peek[0] == 0xFF && peek[1] == 0xFE:
63 return UTF16LE
64 case len(peek) >= 2 && peek[0] == 0xFE && peek[1] == 0xFF:
65 return UTF16BE
66 }
67 return UTF8
68 }
69
70 // DetectUTF16NoBOM heuristically recognises BOM-less UTF-16 from the NUL-byte
71 // distribution: ASCII-range text encodes one byte of payload and one 0x00 per
72 // code unit, so the NULs cluster on odd offsets (LE) or even offsets (BE). It
73 // requires a strong skew — one parity heavily NUL, the other almost none — so
74 // genuine binary (NULs on both parities) and plain UTF-8 (no NULs) fall through.
75 func DetectUTF16NoBOM(b []byte) (Kind, bool) {
76 n := len(b)
77 if n < 16 {
78 return UTF8, false
79 }
80 n &^= 1 // examine an even-length window so parity counts are comparable
81 var evenNUL, oddNUL int
82 for i := range n {
83 if b[i] != 0 {
84 continue
85 }
86 if i%2 == 0 {
87 evenNUL++
88 } else {
89 oddNUL++
90 }
91 }
92 half := n / 2
93 switch {
94 case oddNUL*10 >= half*3 && evenNUL*20 <= half:
95 return UTF16LENoBOM, true
96 case evenNUL*10 >= half*3 && oddNUL*20 <= half:
97 return UTF16BENoBOM, true
98 }
99 return UTF8, false
100 }
101
102 // Decode converts data from the given encoding to UTF-8 bytes.
103 func Decode(data []byte, enc Kind) []byte {
104 switch enc {
105 case UTF8BOM:
106 return data[3:]
107 case UTF16LE:
108 return decodeUTF16(data[2:], binary.LittleEndian)
109 case UTF16BE:
110 return decodeUTF16(data[2:], binary.BigEndian)
111 case UTF16LENoBOM:
112 return decodeUTF16(data, binary.LittleEndian)
113 case UTF16BENoBOM:
114 return decodeUTF16(data, binary.BigEndian)
115 case GB18030, GBK:
116 out, _, err := transform.Bytes(charsetOf(enc).NewDecoder(), data)
117 if err != nil {
118 return data // should not happen after Detect, but be safe
119 }
120 return out
121 }
122 // UTF8 and LossyUTF8 both pass through — LossyUTF8 is already
123 // "best effort" and Go strings can hold arbitrary bytes.
124 return data
125 }
126
127 // DecodeToUTF8 converts raw text-like file bytes to UTF-8 using Reasonix's
128 // shared detection cascade. It is intended for user-editable structured files
129 // (TOML, JSON, dotenv, Markdown) before handing the content to strict parsers.
130 func DecodeToUTF8(data []byte) []byte {
131 _, out := DetectAndDecode(data)
132 return out
133 }
134
135 // ReadFileUTF8 reads path and decodes text-like content to UTF-8.
136 func ReadFileUTF8(path string) ([]byte, error) {
137 data, err := os.ReadFile(path)
138 if err != nil {
139 return nil, err
140 }
141 return DecodeToUTF8(data), nil
142 }
143
144 // Decoder returns a streaming transform.Transformer for the given encoding,
145 // suitable for wrapping an io.Reader via dec.Reader(r). Returns nil for UTF-8
146 // and LossyUTF8 (no transformation needed — the caller should read directly).
147 func Decoder(enc Kind) transform.Transformer {
148 switch enc {
149 case UTF8BOM:
150 // UTF-8 BOM just needs the 3-byte prefix stripped; the content is
151 // already valid UTF-8. Callers handle BOM stripping via Decode.
152 return nil
153 case GB18030, GBK:
154 return charsetOf(enc).NewDecoder()
155 case UTF16LE:
156 return unicode.UTF16(unicode.LittleEndian, unicode.ExpectBOM).NewDecoder()
157 case UTF16BE:
158 return unicode.UTF16(unicode.BigEndian, unicode.ExpectBOM).NewDecoder()
159 case UTF16LENoBOM:
160 return unicode.UTF16(unicode.LittleEndian, unicode.IgnoreBOM).NewDecoder()
161 case UTF16BENoBOM:
162 return unicode.UTF16(unicode.BigEndian, unicode.IgnoreBOM).NewDecoder()
163 }
164 // UTF8 and LossyUTF8 need no transformation.
165 return nil
166 }
167
168 // Encode converts a UTF-8 string back to the given file encoding. UTF8 and
169 // LossyUTF8 produce plain UTF-8 bytes. A legacy charset that cannot represent a
170 // character answers ErrUnencodable rather than switching the file to UTF-8.
171 func Encode(text string, enc Kind) ([]byte, error) {
172 switch enc {
173 case UTF8BOM:
174 return append(utf8BOM, []byte(text)...), nil
175 case UTF16LE:
176 return encodeUTF16(text, binary.LittleEndian, true), nil
177 case UTF16BE:
178 return encodeUTF16(text, binary.BigEndian, true), nil
179 case UTF16LENoBOM:
180 return encodeUTF16(text, binary.LittleEndian, false), nil
181 case UTF16BENoBOM:
182 return encodeUTF16(text, binary.BigEndian, false), nil
183 case GB18030, GBK:
184 return encodeCharset(text, enc)
185 }
186 return []byte(text), nil
187 }
188
189 // MustEncode is Encode for text known to be representable, such as a fixture.
190 func MustEncode(text string, enc Kind) []byte {
191 out, err := Encode(text, enc)
192 if err != nil {
193 panic(err)
194 }
195 return out
196 }
197
198 // decodeUTF16 converts UTF-16 bytes (BOM already stripped) to UTF-8.
199 func decodeUTF16(b []byte, order binary.ByteOrder) []byte {
200 u := make([]uint16, 0, len(b)/2)
201 for i := 0; i+1 < len(b); i += 2 {
202 u = append(u, order.Uint16(b[i:i+2]))
203 }
204 return []byte(string(utf16Decode(u)))
205 }
206
207 // encodeUTF16 converts a UTF-8 string to UTF-16 bytes, with a BOM when withBOM.
208 func encodeUTF16(text string, order binary.ByteOrder, withBOM bool) []byte {
209 runes := []rune(text)
210 encoded := utf16Encode(runes)
211
212 var buf bytes.Buffer
213 if withBOM {
214 var bom [2]byte
215 if order == binary.LittleEndian {
216 bom[0], bom[1] = 0xFF, 0xFE
217 } else {
218 bom[0], bom[1] = 0xFE, 0xFF
219 }
220 buf.Write(bom[:])
221 }
222 for _, u := range encoded {
223 var b [2]byte
224 order.PutUint16(b[:], u)
225 buf.Write(b[:])
226 }
227 return buf.Bytes()
228 }
229
230 // utf16Decode converts UTF-16 code units to runes, handling surrogate pairs.
231 func utf16Decode(u []uint16) []rune {
232 var out []rune
233 for i := 0; i < len(u); i++ {
234 c := u[i]
235 if c >= 0xD800 && c <= 0xDBFF && i+1 < len(u) {
236 c2 := u[i+1]
237 if c2 >= 0xDC00 && c2 <= 0xDFFF {
238 out = append(out, rune(c-0xD800)<<10|rune(c2-0xDC00)+0x10000)
239 i++
240 continue
241 }
242 }
243 out = append(out, rune(c))
244 }
245 return out
246 }
247
248 // utf16Encode converts runes to UTF-16 code units, producing surrogates for
249 // supplementary plane characters.
250 func utf16Encode(runes []rune) []uint16 {
251 var out []uint16
252 for _, r := range runes {
253 if r >= 0x10000 && r <= 0x10FFFF {
254 r -= 0x10000
255 out = append(out, uint16(0xD800+(r>>10)), uint16(0xDC00+(r&0x3FF)))
256 } else {
257 out = append(out, uint16(r))
258 }
259 }
260 return out
261 }
262
262 lines GO