| 1 | package encoding |
| 2 | |
| 3 | import ( |
| 4 | "bytes" |
| 5 | "errors" |
| 6 | "strings" |
| 7 | "testing" |
| 8 | |
| 9 | "golang.org/x/text/encoding/simplifiedchinese" |
| 10 | ) |
| 11 | |
| 12 | // CP936 writes the euro as the single byte 0x80, which GB18030 cannot restore. |
| 13 | // Such a file is GBK: it keeps its bytes and an edit writes CP936 back. |
| 14 | func TestDetectGBKWithEuroByte(t *testing.T) { |
| 15 | data := []byte{0xd6, 0xd0, 0xce, 0xc4, 0x80} |
| 16 | enc, _ := Detect(data) |
| 17 | if enc != GBK { |
| 18 | t.Fatalf("got %v, want GBK", enc) |
| 19 | } |
| 20 | if got := string(Decode(data, enc)); got != "中文€" { |
| 21 | t.Fatalf("Decode = %q, want %q", got, "中文€") |
| 22 | } |
| 23 | if out := MustEncode("中文€", enc); !bytes.Equal(out, data) { |
| 24 | t.Fatalf("Encode = % x, want % x", out, data) |
| 25 | } |
| 26 | } |
| 27 | |
| 28 | // A fragment cut inside a character, with no newline to cut at instead, is |
| 29 | // still the charset it was written in; the split character is left out. |
| 30 | func TestDetectFragmentDropsTheSplitCharacter(t *testing.T) { |
| 31 | gb, err := simplifiedchinese.GB18030.NewEncoder().String("x" + strings.Repeat("啊", 100)) |
| 32 | if err != nil { |
| 33 | t.Fatal(err) |
| 34 | } |
| 35 | cut := []byte(gb)[:len(gb)-1] |
| 36 | enc, whole := DetectFragment(cut) |
| 37 | if enc != GB18030 { |
| 38 | t.Fatalf("got %v, want GB18030", enc) |
| 39 | } |
| 40 | if len(whole) != len(cut)-1 { |
| 41 | t.Fatalf("whole characters = %d bytes, want %d", len(whole), len(cut)-1) |
| 42 | } |
| 43 | utf := []byte("x" + strings.Repeat("啊", 100)) |
| 44 | if enc, whole := DetectFragment(utf[:len(utf)-1]); enc != UTF8 || len(whole) != len(utf)-3 { |
| 45 | t.Fatalf("UTF-8 fragment = %v, %d bytes; want UTF8, %d", enc, len(whole), len(utf)-3) |
| 46 | } |
| 47 | } |
| 48 | |
| 49 | // Output that no charset restores, because a buffer cut split a code-page |
| 50 | // character, is still read in the code page. |
| 51 | func TestDecodeOutputReadsCutCodePageText(t *testing.T) { |
| 52 | gb, _ := simplifiedchinese.GB18030.NewEncoder().String("'sh' 不是内部或外部命令") |
| 53 | if got := string(DecodeOutput([]byte(gb)[:len(gb)-1], Cut{Tail: true})); !strings.HasPrefix(got, "'sh' 不是内部或外部命") { |
| 54 | t.Fatalf("DecodeOutput = %q", got) |
| 55 | } |
| 56 | } |
| 57 | |
| 58 | // GBK cannot hold every character. Encode names the first one it cannot hold |
| 59 | // and where it sits, rather than writing the text as UTF-8. |
| 60 | func TestEncodeRefusesCharacterTheCharsetCannotHold(t *testing.T) { |
| 61 | for _, r := range []string{"✅", "🚀", "𠀀", "™", "ᠠ"} { |
| 62 | _, err := Encode("中文"+r, GBK) |
| 63 | var ue *UnencodableError |
| 64 | if !errors.Is(err, ErrUnencodable) || !errors.As(err, &ue) { |
| 65 | t.Fatalf("Encode(%q, GBK) err = %v, want ErrUnencodable", r, err) |
| 66 | } |
| 67 | if string(ue.Rune) != r || ue.Offset != len("中文") || ue.Charset != "GBK" { |
| 68 | t.Fatalf("UnencodableError = %+v, want %q at byte %d in GBK", ue, r, len("中文")) |
| 69 | } |
| 70 | if _, err := Encode("中文"+r, GB18030); err != nil { |
| 71 | t.Fatalf("GB18030 holds every character, got %v for %q", err, r) |
| 72 | } |
| 73 | } |
| 74 | } |
| 75 | |
| 76 | // Process output cut inside a UTF-8 character at either end is read as UTF-8, |
| 77 | // less the cut character, not reinterpreted as a code page. |
| 78 | func TestDecodeOutputTrimsCutUTF8Edges(t *testing.T) { |
| 79 | full := []byte("参数格式不正确") |
| 80 | if got := string(DecodeOutput(full[1:], Cut{Head: true})); got != "数格式不正确" { |
| 81 | t.Fatalf("front cut = %q", got) |
| 82 | } |
| 83 | if got := string(DecodeOutput(full[:len(full)-1], Cut{Tail: true})); got != "参数格式不正" { |
| 84 | t.Fatalf("back cut = %q", got) |
| 85 | } |
| 86 | if got := string(DecodeOutput(full[2:len(full)-2], Cut{Head: true, Tail: true})); got != "数格式不正" { |
| 87 | t.Fatalf("both cut = %q", got) |
| 88 | } |
| 89 | } |
| 90 | |
| 91 | // An end no bound cut is where the output began or ended, so a code-page pair |
| 92 | // there is kept rather than trimmed as half of a UTF-8 character. |
| 93 | func TestDecodeOutputKeepsUncutCodePageEdges(t *testing.T) { |
| 94 | cases := map[string]string{ |
| 95 | "\xbc\xdb": "价", |
| 96 | "\xb0\xa1\r\n": "啊\r\n", |
| 97 | "ok \xe4\xa1": "ok 洹", |
| 98 | "\xb0\xa1 ok": "啊 ok", |
| 99 | } |
| 100 | for raw, want := range cases { |
| 101 | if got := string(DecodeOutput([]byte(raw), Cut{})); got != want { |
| 102 | t.Fatalf("DecodeOutput(% x) = %q, want %q", raw, got, want) |
| 103 | } |
| 104 | } |
| 105 | } |
| 106 | |
| 107 | // A process stopped mid-write leaves one incomplete UTF-8 character at an end |
| 108 | // nothing cut. Once a whole multi-byte UTF-8 character has appeared, that is |
| 109 | // the only invalid part, and the output is still UTF-8. |
| 110 | func TestDecodeOutputReadsUncutUTF8EndingInPartialRune(t *testing.T) { |
| 111 | const text = "编译成功…正在运行" |
| 112 | half := []byte("中")[:2] |
| 113 | if got := string(DecodeOutput(append([]byte(text), half...), Cut{})); got != text { |
| 114 | t.Fatalf("UTF-8 plus half a rune = %q, want %q", got, text) |
| 115 | } |
| 116 | // Nothing multi-byte came before the half rune, so no structure says UTF-8; |
| 117 | // the bytes still form a GBK pair and are read as one. |
| 118 | if got := string(DecodeOutput(append([]byte("ok "), half...), Cut{})); got != "ok 涓" { |
| 119 | t.Fatalf("ASCII plus half a rune = %q", got) |
| 120 | } |
| 121 | gb, _ := simplifiedchinese.GB18030.NewEncoder().String("参数格式不正确") |
| 122 | for _, tc := range []struct { |
| 123 | name string |
| 124 | raw string |
| 125 | cut Cut |
| 126 | want string |
| 127 | }{ |
| 128 | {"uncut GBK", gb, Cut{}, "参数格式不正确"}, |
| 129 | {"GBK cut at the end", gb[:len(gb)-1], Cut{Tail: true}, "参数格式不正"}, |
| 130 | {"GBK cut at the start", gb[2:], Cut{Head: true}, "数格式不正确"}, |
| 131 | } { |
| 132 | got := string(DecodeOutput([]byte(tc.raw), tc.cut)) |
| 133 | if got != tc.want && !(tc.cut.Tail && strings.HasPrefix(got, tc.want)) { |
| 134 | t.Fatalf("%s = %q, want %q", tc.name, got, tc.want) |
| 135 | } |
| 136 | } |
| 137 | } |
| 138 |