返回 DeepSeek-Reasonix
charset.go
根目录 / internal / fileutil / encoding / charset.go
1 package encoding
2
3 import (
4 "bytes"
5 "errors"
6 "fmt"
7 "unicode/utf8"
8
9 "golang.org/x/text/encoding"
10 "golang.org/x/text/encoding/simplifiedchinese"
11 "golang.org/x/text/transform"
12 )
13
14 // charsets are the legacy encodings a file may be stored in, in preference
15 // order. GBK comes second because it differs from GB18030 only where GB18030
16 // cannot restore the bytes, such as CP936's single-byte euro, 0x80.
17 var charsets = []struct {
18 kind Kind
19 name string
20 enc encoding.Encoding
21 }{
22 {GB18030, "GB18030", simplifiedchinese.GB18030},
23 {GBK, "GBK", simplifiedchinese.GBK},
24 }
25
26 func charsetOf(k Kind) encoding.Encoding {
27 for _, c := range charsets {
28 if c.kind == k {
29 return c.enc
30 }
31 }
32 return nil
33 }
34
35 // ErrUnencodable is the identity of a write holding a character the file's
36 // encoding cannot represent.
37 var ErrUnencodable = errors.New("character not representable in the file's encoding")
38
39 // UnencodableError names the first character a legacy charset cannot represent.
40 type UnencodableError struct {
41 Charset string
42 Rune rune
43 Offset int // byte offset of Rune in the text being written
44 }
45
46 func (e *UnencodableError) Error() string {
47 return fmt.Sprintf("%q (U+%04X) at byte %d cannot be written in %s, the file's encoding", e.Rune, e.Rune, e.Offset, e.Charset)
48 }
49
50 func (e *UnencodableError) Unwrap() error { return ErrUnencodable }
51
52 func encodeCharset(text string, k Kind) ([]byte, error) {
53 for _, c := range charsets {
54 if c.kind != k {
55 continue
56 }
57 out, n, err := transform.Bytes(c.enc.NewEncoder(), []byte(text))
58 if err != nil {
59 r, _ := utf8.DecodeRuneInString(text[n:])
60 return nil, &UnencodableError{Charset: c.name, Rune: r, Offset: n}
61 }
62 return out, nil
63 }
64 return []byte(text), nil
65 }
66
67 // DetectFragment is Detect for data that is only the start of a longer stream:
68 // a character the cut split is dropped rather than read as proof against the
69 // encoding. It returns the prefix that holds whole characters.
70 func DetectFragment(data []byte) (Kind, []byte) {
71 k, n, _ := sniff(data, false)
72 return k, data[:n]
73 }
74
75 // DetectAndDecode is Detect followed by Decode, decoding a legacy charset once.
76 func DetectAndDecode(data []byte) (Kind, []byte) {
77 k, _, text := sniff(data, true)
78 if text != nil {
79 return k, text
80 }
81 return k, Decode(data, k)
82 }
83
84 // Cut says which ends of a bounded buffer lost bytes to its bound. Only a cut
85 // end can hold part of a character; an uncut end is where the output began or
86 // ended, so a byte there belongs to it, such as half of a GBK pair.
87 type Cut struct {
88 Head bool // bytes before the buffer were dropped
89 Tail bool // bytes after the buffer were dropped
90 }
91
92 // DecodeOutput reads bytes that are only displayed, never written back, such as
93 // a process's output. UTF-8 is read without a character split at an end cut
94 // says was cut; what no charset restores is still read as GB18030, since a cut
95 // can split a code-page character anywhere.
96 func DecodeOutput(data []byte, cut Cut) []byte {
97 edges := cut.trim(data)
98 if utf8.Valid(edges) {
99 return edges
100 }
101 // A process stopped mid-write ends inside a character no bound cut. That is
102 // still UTF-8 when it is the only invalid part and a multi-byte UTF-8
103 // character already appeared; ASCII alone proves nothing.
104 if whole := trimPartialRune(edges); len(whole) < len(edges) && utf8.Valid(whole) && utf8.RuneCount(whole) < len(whole) {
105 return whole
106 }
107 k, _, text := sniff(data, true)
108 if text != nil {
109 return text
110 }
111 if k == LossyUTF8 {
112 k = GB18030
113 }
114 return Decode(data, k)
115 }
116
117 // sniff detects data's encoding. When final is false data is a fragment, and n
118 // excludes a trailing sequence it cut short. text is the decoded data[:n] when
119 // a legacy charset won, so the caller need not decode it again.
120 func sniff(data []byte, final bool) (k Kind, n int, text []byte) {
121 switch {
122 case len(data) >= 3 && data[0] == 0xEF && data[1] == 0xBB && data[2] == 0xBF:
123 return UTF8BOM, len(data), nil
124 case len(data) >= 2 && data[0] == 0xFF && data[1] == 0xFE:
125 return UTF16LE, len(data), nil
126 case len(data) >= 2 && data[0] == 0xFE && data[1] == 0xFF:
127 return UTF16BE, len(data), nil
128 }
129 // BOM-less UTF-16 must be tried before utf8.Valid: its low bytes plus 0x00
130 // high bytes are all valid UTF-8 code units.
131 if k, ok := DetectUTF16NoBOM(data); ok {
132 return k, len(data), nil
133 }
134 whole := data
135 if !final {
136 whole = trimPartialRune(data)
137 }
138 if utf8.Valid(whole) {
139 return UTF8, len(whole), nil
140 }
141 for _, c := range charsets {
142 if text, n, ok := roundTrip(c.enc, data, final); ok {
143 return c.kind, n, text
144 }
145 }
146 return LossyUTF8, len(data), nil
147 }
148
149 // roundTrip decodes data and reports whether encoding the text restores it byte
150 // for byte. The decoders never fail: they turn an invalid sequence into U+FFFD,
151 // which encodes back as different bytes, so decoding alone is no signal.
152 func roundTrip(e encoding.Encoding, data []byte, final bool) ([]byte, int, bool) {
153 var text []byte
154 n := len(data)
155 if final {
156 var err error
157 if text, _, err = transform.Bytes(e.NewDecoder(), data); err != nil {
158 return nil, 0, false
159 }
160 } else {
161 // Not at EOF, the decoder stops before an incomplete trailing sequence
162 // with ErrShortSrc; nSrc is then the prefix of whole characters. One
163 // source byte decodes to at most three.
164 dst := make([]byte, 3*len(data)+utf8.UTFMax)
165 nDst, nSrc, err := e.NewDecoder().Transform(dst, data, false)
166 if err != nil && !errors.Is(err, transform.ErrShortSrc) {
167 return nil, 0, false
168 }
169 text, n = dst[:nDst], nSrc
170 }
171 back, _, err := transform.Bytes(e.NewEncoder(), text)
172 if err != nil || !bytes.Equal(back, data[:n]) {
173 return nil, 0, false
174 }
175 return text, n, true
176 }
177
178 // trimPartialRune drops a trailing UTF-8 sequence the cut left incomplete. A
179 // byte that is simply invalid stays, and still fails the validity check.
180 func trimPartialRune(data []byte) []byte {
181 for i := len(data) - 1; i >= 0 && len(data)-i < utf8.UTFMax; i-- {
182 if !utf8.RuneStart(data[i]) {
183 continue
184 }
185 if utf8.FullRune(data[i:]) {
186 return data
187 }
188 return data[:i]
189 }
190 return data
191 }
192
193 // trim drops the continuation bytes a head cut left at the front and the
194 // sequence a tail cut left incomplete at the back.
195 func (c Cut) trim(data []byte) []byte {
196 for i := 0; c.Head && i < utf8.UTFMax-1 && len(data) > 0 && !utf8.RuneStart(data[0]); i++ {
197 data = data[1:]
198 }
199 if c.Tail {
200 data = trimPartialRune(data)
201 }
202 return data
203 }
204
204 lines GO