返回 DeepSeek-Reasonix
charset_test.go
根目录 / internal / fileutil / encoding / charset_test.go
1 package encoding
2
3 import (
4 "bytes"
5 "errors"
6 "strings"
7 "testing"
8
9 "golang.org/x/text/encoding/simplifiedchinese"
10 )
11
12 // CP936 writes the euro as the single byte 0x80, which GB18030 cannot restore.
13 // Such a file is GBK: it keeps its bytes and an edit writes CP936 back.
14 func TestDetectGBKWithEuroByte(t *testing.T) {
15 data := []byte{0xd6, 0xd0, 0xce, 0xc4, 0x80}
16 enc, _ := Detect(data)
17 if enc != GBK {
18 t.Fatalf("got %v, want GBK", enc)
19 }
20 if got := string(Decode(data, enc)); got != "中文€" {
21 t.Fatalf("Decode = %q, want %q", got, "中文€")
22 }
23 if out := MustEncode("中文€", enc); !bytes.Equal(out, data) {
24 t.Fatalf("Encode = % x, want % x", out, data)
25 }
26 }
27
28 // A fragment cut inside a character, with no newline to cut at instead, is
29 // still the charset it was written in; the split character is left out.
30 func TestDetectFragmentDropsTheSplitCharacter(t *testing.T) {
31 gb, err := simplifiedchinese.GB18030.NewEncoder().String("x" + strings.Repeat("啊", 100))
32 if err != nil {
33 t.Fatal(err)
34 }
35 cut := []byte(gb)[:len(gb)-1]
36 enc, whole := DetectFragment(cut)
37 if enc != GB18030 {
38 t.Fatalf("got %v, want GB18030", enc)
39 }
40 if len(whole) != len(cut)-1 {
41 t.Fatalf("whole characters = %d bytes, want %d", len(whole), len(cut)-1)
42 }
43 utf := []byte("x" + strings.Repeat("啊", 100))
44 if enc, whole := DetectFragment(utf[:len(utf)-1]); enc != UTF8 || len(whole) != len(utf)-3 {
45 t.Fatalf("UTF-8 fragment = %v, %d bytes; want UTF8, %d", enc, len(whole), len(utf)-3)
46 }
47 }
48
49 // Output that no charset restores, because a buffer cut split a code-page
50 // character, is still read in the code page.
51 func TestDecodeOutputReadsCutCodePageText(t *testing.T) {
52 gb, _ := simplifiedchinese.GB18030.NewEncoder().String("'sh' 不是内部或外部命令")
53 if got := string(DecodeOutput([]byte(gb)[:len(gb)-1], Cut{Tail: true})); !strings.HasPrefix(got, "'sh' 不是内部或外部命") {
54 t.Fatalf("DecodeOutput = %q", got)
55 }
56 }
57
58 // GBK cannot hold every character. Encode names the first one it cannot hold
59 // and where it sits, rather than writing the text as UTF-8.
60 func TestEncodeRefusesCharacterTheCharsetCannotHold(t *testing.T) {
61 for _, r := range []string{"✅", "🚀", "𠀀", "™", "ᠠ"} {
62 _, err := Encode("中文"+r, GBK)
63 var ue *UnencodableError
64 if !errors.Is(err, ErrUnencodable) || !errors.As(err, &ue) {
65 t.Fatalf("Encode(%q, GBK) err = %v, want ErrUnencodable", r, err)
66 }
67 if string(ue.Rune) != r || ue.Offset != len("中文") || ue.Charset != "GBK" {
68 t.Fatalf("UnencodableError = %+v, want %q at byte %d in GBK", ue, r, len("中文"))
69 }
70 if _, err := Encode("中文"+r, GB18030); err != nil {
71 t.Fatalf("GB18030 holds every character, got %v for %q", err, r)
72 }
73 }
74 }
75
76 // Process output cut inside a UTF-8 character at either end is read as UTF-8,
77 // less the cut character, not reinterpreted as a code page.
78 func TestDecodeOutputTrimsCutUTF8Edges(t *testing.T) {
79 full := []byte("参数格式不正确")
80 if got := string(DecodeOutput(full[1:], Cut{Head: true})); got != "数格式不正确" {
81 t.Fatalf("front cut = %q", got)
82 }
83 if got := string(DecodeOutput(full[:len(full)-1], Cut{Tail: true})); got != "参数格式不正" {
84 t.Fatalf("back cut = %q", got)
85 }
86 if got := string(DecodeOutput(full[2:len(full)-2], Cut{Head: true, Tail: true})); got != "数格式不正" {
87 t.Fatalf("both cut = %q", got)
88 }
89 }
90
91 // An end no bound cut is where the output began or ended, so a code-page pair
92 // there is kept rather than trimmed as half of a UTF-8 character.
93 func TestDecodeOutputKeepsUncutCodePageEdges(t *testing.T) {
94 cases := map[string]string{
95 "\xbc\xdb": "价",
96 "\xb0\xa1\r\n": "啊\r\n",
97 "ok \xe4\xa1": "ok 洹",
98 "\xb0\xa1 ok": "啊 ok",
99 }
100 for raw, want := range cases {
101 if got := string(DecodeOutput([]byte(raw), Cut{})); got != want {
102 t.Fatalf("DecodeOutput(% x) = %q, want %q", raw, got, want)
103 }
104 }
105 }
106
107 // A process stopped mid-write leaves one incomplete UTF-8 character at an end
108 // nothing cut. Once a whole multi-byte UTF-8 character has appeared, that is
109 // the only invalid part, and the output is still UTF-8.
110 func TestDecodeOutputReadsUncutUTF8EndingInPartialRune(t *testing.T) {
111 const text = "编译成功…正在运行"
112 half := []byte("中")[:2]
113 if got := string(DecodeOutput(append([]byte(text), half...), Cut{})); got != text {
114 t.Fatalf("UTF-8 plus half a rune = %q, want %q", got, text)
115 }
116 // Nothing multi-byte came before the half rune, so no structure says UTF-8;
117 // the bytes still form a GBK pair and are read as one.
118 if got := string(DecodeOutput(append([]byte("ok "), half...), Cut{})); got != "ok 涓" {
119 t.Fatalf("ASCII plus half a rune = %q", got)
120 }
121 gb, _ := simplifiedchinese.GB18030.NewEncoder().String("参数格式不正确")
122 for _, tc := range []struct {
123 name string
124 raw string
125 cut Cut
126 want string
127 }{
128 {"uncut GBK", gb, Cut{}, "参数格式不正确"},
129 {"GBK cut at the end", gb[:len(gb)-1], Cut{Tail: true}, "参数格式不正"},
130 {"GBK cut at the start", gb[2:], Cut{Head: true}, "数格式不正确"},
131 } {
132 got := string(DecodeOutput([]byte(tc.raw), tc.cut))
133 if got != tc.want && !(tc.cut.Tail && strings.HasPrefix(got, tc.want)) {
134 t.Fatalf("%s = %q, want %q", tc.name, got, tc.want)
135 }
136 }
137 }
138
138 lines GO