| 1 | package encoding |
| 2 | |
| 3 | import ( |
| 4 | "bytes" |
| 5 | "encoding/binary" |
| 6 | "os" |
| 7 | "path/filepath" |
| 8 | "strings" |
| 9 | "testing" |
| 10 | "unicode/utf16" |
| 11 | |
| 12 | "golang.org/x/text/encoding/simplifiedchinese" |
| 13 | ) |
| 14 | |
| 15 | // Detect |
| 16 | |
| 17 | func TestDetectUTF8Plain(t *testing.T) { |
| 18 | enc, _ := Detect([]byte("hello world\n")) |
| 19 | if enc != UTF8 { |
| 20 | t.Errorf("got %v, want UTF8", enc) |
| 21 | } |
| 22 | } |
| 23 | |
| 24 | func TestDetectUTF8BOM(t *testing.T) { |
| 25 | in := append([]byte{0xEF, 0xBB, 0xBF}, []byte("hello")...) |
| 26 | enc, _ := Detect(in) |
| 27 | if enc != UTF8BOM { |
| 28 | t.Errorf("got %v, want UTF8BOM", enc) |
| 29 | } |
| 30 | } |
| 31 | |
| 32 | func TestDetectUTF16LE(t *testing.T) { |
| 33 | var b bytes.Buffer |
| 34 | b.Write([]byte{0xFF, 0xFE}) |
| 35 | for _, r := range utf16.Encode([]rune("hello")) { |
| 36 | _ = binary.Write(&b, binary.LittleEndian, r) |
| 37 | } |
| 38 | enc, _ := Detect(b.Bytes()) |
| 39 | if enc != UTF16LE { |
| 40 | t.Errorf("got %v, want UTF16LE", enc) |
| 41 | } |
| 42 | } |
| 43 | |
| 44 | func TestDetectUTF16BE(t *testing.T) { |
| 45 | var b bytes.Buffer |
| 46 | b.Write([]byte{0xFE, 0xFF}) |
| 47 | for _, r := range utf16.Encode([]rune("hello")) { |
| 48 | _ = binary.Write(&b, binary.BigEndian, r) |
| 49 | } |
| 50 | enc, _ := Detect(b.Bytes()) |
| 51 | if enc != UTF16BE { |
| 52 | t.Errorf("got %v, want UTF16BE", enc) |
| 53 | } |
| 54 | } |
| 55 | |
| 56 | func TestDetectGB18030(t *testing.T) { |
| 57 | gb, err := simplifiedchinese.GB18030.NewEncoder().String("你好世界") |
| 58 | if err != nil { |
| 59 | t.Fatalf("encode: %v", err) |
| 60 | } |
| 61 | enc, _ := Detect([]byte(gb)) |
| 62 | if enc != GB18030 { |
| 63 | t.Errorf("got %v, want GB18030", enc) |
| 64 | } |
| 65 | } |
| 66 | |
| 67 | // The GB18030 decoder accepts any bytes, mapping invalid ones to U+FFFD, so a |
| 68 | // UTF-8 file with one stray byte must not be taken for GB18030: re-encoding it |
| 69 | // would rewrite every character the decoder could not read. |
| 70 | func TestDetectUTF8WithStrayByteIsNotGB18030(t *testing.T) { |
| 71 | data := []byte("// 中文注释\nvalue := 1\n\xff\n") |
| 72 | if enc, _ := Detect(data); enc != LossyUTF8 { |
| 73 | t.Fatalf("got %v, want LossyUTF8", enc) |
| 74 | } |
| 75 | if out := MustEncode(string(Decode(data, LossyUTF8)), LossyUTF8); !bytes.Equal(out, data) { |
| 76 | t.Fatalf("LossyUTF8 round trip changed bytes: % x", out) |
| 77 | } |
| 78 | } |
| 79 | |
| 80 | func TestDetectEmpty(t *testing.T) { |
| 81 | enc, _ := Detect(nil) |
| 82 | if enc != UTF8 { |
| 83 | t.Errorf("empty input: got %v, want UTF8", enc) |
| 84 | } |
| 85 | } |
| 86 | |
| 87 | // Decode |
| 88 | |
| 89 | func TestDecodeUTF8(t *testing.T) { |
| 90 | in := []byte("hello\n世界") |
| 91 | out := Decode(in, UTF8) |
| 92 | if string(out) != "hello\n世界" { |
| 93 | t.Errorf("got %q", out) |
| 94 | } |
| 95 | } |
| 96 | |
| 97 | func TestReadFileUTF8DecodesGB18030(t *testing.T) { |
| 98 | path := filepath.Join(t.TempDir(), "settings.json") |
| 99 | if err := os.WriteFile(path, MustEncode(`{"label":"中文"}`, GB18030), 0o644); err != nil { |
| 100 | t.Fatal(err) |
| 101 | } |
| 102 | got, err := ReadFileUTF8(path) |
| 103 | if err != nil { |
| 104 | t.Fatal(err) |
| 105 | } |
| 106 | if string(got) != `{"label":"中文"}` { |
| 107 | t.Fatalf("ReadFileUTF8 = %q", got) |
| 108 | } |
| 109 | } |
| 110 | |
| 111 | func TestDecodeUTF8BOM(t *testing.T) { |
| 112 | in := append([]byte{0xEF, 0xBB, 0xBF}, []byte("hello")...) |
| 113 | out := Decode(in, UTF8BOM) |
| 114 | if string(out) != "hello" { |
| 115 | t.Errorf("got %q, want 'hello'", out) |
| 116 | } |
| 117 | if bytes.Contains(out, []byte{0xEF, 0xBB, 0xBF}) { |
| 118 | t.Error("BOM leaked into decoded output") |
| 119 | } |
| 120 | } |
| 121 | |
| 122 | func TestDecodeUTF16LE(t *testing.T) { |
| 123 | var b bytes.Buffer |
| 124 | b.Write([]byte{0xFF, 0xFE}) |
| 125 | for _, r := range utf16.Encode([]rune("hello\nworld")) { |
| 126 | _ = binary.Write(&b, binary.LittleEndian, r) |
| 127 | } |
| 128 | out := Decode(b.Bytes(), UTF16LE) |
| 129 | if string(out) != "hello\nworld" { |
| 130 | t.Errorf("got %q", out) |
| 131 | } |
| 132 | } |
| 133 | |
| 134 | func TestDecodeUTF16BE(t *testing.T) { |
| 135 | var b bytes.Buffer |
| 136 | b.Write([]byte{0xFE, 0xFF}) |
| 137 | for _, r := range utf16.Encode([]rune("hello\nworld")) { |
| 138 | _ = binary.Write(&b, binary.BigEndian, r) |
| 139 | } |
| 140 | out := Decode(b.Bytes(), UTF16BE) |
| 141 | if string(out) != "hello\nworld" { |
| 142 | t.Errorf("got %q", out) |
| 143 | } |
| 144 | } |
| 145 | |
| 146 | func TestDecodeGB18030(t *testing.T) { |
| 147 | gb, _ := simplifiedchinese.GB18030.NewEncoder().String("你好世界\n第二行") |
| 148 | out := Decode([]byte(gb), GB18030) |
| 149 | if string(out) != "你好世界\n第二行" { |
| 150 | t.Errorf("got %q", out) |
| 151 | } |
| 152 | } |
| 153 | |
| 154 | func TestDecodeLossyUTF8(t *testing.T) { |
| 155 | in := []byte{0xFF, 0xFE, 'a'} |
| 156 | out := Decode(in, LossyUTF8) |
| 157 | if !bytes.Equal(out, in) { |
| 158 | t.Errorf("LossyUTF8 should pass through, got %q", out) |
| 159 | } |
| 160 | } |
| 161 | |
| 162 | // Encode |
| 163 | |
| 164 | func TestEncodeUTF8(t *testing.T) { |
| 165 | out := MustEncode("hello", UTF8) |
| 166 | if string(out) != "hello" { |
| 167 | t.Errorf("got %q", out) |
| 168 | } |
| 169 | } |
| 170 | |
| 171 | func TestEncodeUTF8BOM(t *testing.T) { |
| 172 | out := MustEncode("hello", UTF8BOM) |
| 173 | if !bytes.HasPrefix(out, []byte{0xEF, 0xBB, 0xBF}) { |
| 174 | t.Error("missing UTF-8 BOM prefix") |
| 175 | } |
| 176 | if string(out[3:]) != "hello" { |
| 177 | t.Errorf("body = %q", out[3:]) |
| 178 | } |
| 179 | } |
| 180 | |
| 181 | func TestEncodeUTF16LE(t *testing.T) { |
| 182 | out := MustEncode("hi", UTF16LE) |
| 183 | if len(out) < 2 || out[0] != 0xFF || out[1] != 0xFE { |
| 184 | t.Error("missing UTF-16LE BOM") |
| 185 | } |
| 186 | decoded := Decode(out, UTF16LE) |
| 187 | if string(decoded) != "hi" { |
| 188 | t.Errorf("round-trip failed: got %q", decoded) |
| 189 | } |
| 190 | } |
| 191 | |
| 192 | func TestEncodeUTF16BE(t *testing.T) { |
| 193 | out := MustEncode("hi", UTF16BE) |
| 194 | if len(out) < 2 || out[0] != 0xFE || out[1] != 0xFF { |
| 195 | t.Error("missing UTF-16BE BOM") |
| 196 | } |
| 197 | decoded := Decode(out, UTF16BE) |
| 198 | if string(decoded) != "hi" { |
| 199 | t.Errorf("round-trip failed: got %q", decoded) |
| 200 | } |
| 201 | } |
| 202 | |
| 203 | func TestEncodeGB18030(t *testing.T) { |
| 204 | out := MustEncode("你好", GB18030) |
| 205 | dec, _ := simplifiedchinese.GB18030.NewDecoder().Bytes(out) |
| 206 | if string(dec) != "你好" { |
| 207 | t.Errorf("round-trip failed: got %q", dec) |
| 208 | } |
| 209 | } |
| 210 | |
| 211 | // Round-trip |
| 212 | |
| 213 | func TestRoundTripGB18030(t *testing.T) { |
| 214 | original := "你好世界\n第二行\n" |
| 215 | gb, _ := simplifiedchinese.GB18030.NewEncoder().String(original) |
| 216 | |
| 217 | enc, _ := Detect([]byte(gb)) |
| 218 | decoded := string(Decode([]byte(gb), enc)) |
| 219 | if decoded != original { |
| 220 | t.Fatalf("decode mismatch: %q", decoded) |
| 221 | } |
| 222 | |
| 223 | edited := strings.Replace(decoded, "第二行", "新的行", 1) |
| 224 | reencoded := MustEncode(edited, enc) |
| 225 | redecoded := string(Decode(reencoded, enc)) |
| 226 | if redecoded != edited { |
| 227 | t.Errorf("round-trip failed: got %q, want %q", redecoded, edited) |
| 228 | } |
| 229 | } |
| 230 | |
| 231 | func TestRoundTripUTF16LE(t *testing.T) { |
| 232 | original := "hello\nworld\n" |
| 233 | encoded := MustEncode(original, UTF16LE) |
| 234 | enc, _ := Detect(encoded) |
| 235 | decoded := string(Decode(encoded, enc)) |
| 236 | if decoded != original { |
| 237 | t.Errorf("round-trip failed: got %q, want %q", decoded, original) |
| 238 | } |
| 239 | } |
| 240 | |
| 241 | func TestRoundTripUTF8BOM(t *testing.T) { |
| 242 | original := "hello world\n" |
| 243 | encoded := MustEncode(original, UTF8BOM) |
| 244 | enc, _ := Detect(encoded) |
| 245 | decoded := string(Decode(encoded, enc)) |
| 246 | if decoded != original { |
| 247 | t.Errorf("round-trip failed: got %q, want %q", decoded, original) |
| 248 | } |
| 249 | } |
| 250 | |
| 251 | // BOM-less UTF-16 |
| 252 | |
| 253 | func utf16NoBOM(t *testing.T, s string, order binary.ByteOrder) []byte { |
| 254 | t.Helper() |
| 255 | var b bytes.Buffer |
| 256 | for _, r := range utf16.Encode([]rune(s)) { |
| 257 | _ = binary.Write(&b, order, r) |
| 258 | } |
| 259 | return b.Bytes() |
| 260 | } |
| 261 | |
| 262 | func TestDetectUTF16LENoBOM(t *testing.T) { |
| 263 | in := utf16NoBOM(t, "// Created by 69431 on 2024/12/31\n#include \"x.h\"\n", binary.LittleEndian) |
| 264 | enc, _ := Detect(in) |
| 265 | if enc != UTF16LENoBOM { |
| 266 | t.Errorf("got %v, want UTF16LENoBOM", enc) |
| 267 | } |
| 268 | } |
| 269 | |
| 270 | func TestDetectUTF16BENoBOM(t *testing.T) { |
| 271 | in := utf16NoBOM(t, "package main\nfunc main() {}\n", binary.BigEndian) |
| 272 | enc, _ := Detect(in) |
| 273 | if enc != UTF16BENoBOM { |
| 274 | t.Errorf("got %v, want UTF16BENoBOM", enc) |
| 275 | } |
| 276 | } |
| 277 | |
| 278 | func TestDetectPlainUTF8NotUTF16(t *testing.T) { |
| 279 | enc, _ := Detect([]byte("the quick brown fox jumps over the lazy dog\n")) |
| 280 | if enc != UTF8 { |
| 281 | t.Errorf("ASCII text misdetected as %v", enc) |
| 282 | } |
| 283 | } |
| 284 | |
| 285 | func TestDetectBinaryNotUTF16NoBOM(t *testing.T) { |
| 286 | // NULs on both parities — genuine binary must not look like BOM-less UTF-16. |
| 287 | bin := []byte{0x00, 0x00, 0x01, 0x00, 0x00, 0x02, 0xFF, 0x00, 0x00, 0x03, 0x00, 0x00, 0x04, 0x00, 0x00, 0x05, 0x00, 0x00} |
| 288 | if k, ok := DetectUTF16NoBOM(bin); ok { |
| 289 | t.Errorf("binary misdetected as %v", k) |
| 290 | } |
| 291 | } |
| 292 | |
| 293 | func TestDecodeUTF16LENoBOM(t *testing.T) { |
| 294 | in := utf16NoBOM(t, "hello\nworld", binary.LittleEndian) |
| 295 | if out := string(Decode(in, UTF16LENoBOM)); out != "hello\nworld" { |
| 296 | t.Errorf("got %q", out) |
| 297 | } |
| 298 | } |
| 299 | |
| 300 | func TestRoundTripUTF16LENoBOM(t *testing.T) { |
| 301 | original := "// c++ source\nint main() { return 0; }\n" |
| 302 | encoded := utf16NoBOM(t, original, binary.LittleEndian) |
| 303 | |
| 304 | enc, _ := Detect(encoded) |
| 305 | if enc != UTF16LENoBOM { |
| 306 | t.Fatalf("detect: got %v", enc) |
| 307 | } |
| 308 | decoded := string(Decode(encoded, enc)) |
| 309 | if decoded != original { |
| 310 | t.Fatalf("decode mismatch: %q", decoded) |
| 311 | } |
| 312 | edited := strings.Replace(decoded, "return 0", "return 1", 1) |
| 313 | reencoded := MustEncode(edited, enc) |
| 314 | if bytes.HasPrefix(reencoded, []byte{0xFF, 0xFE}) || bytes.HasPrefix(reencoded, []byte{0xFE, 0xFF}) { |
| 315 | t.Error("no-BOM re-encode leaked a BOM") |
| 316 | } |
| 317 | if redecoded := string(Decode(reencoded, enc)); redecoded != edited { |
| 318 | t.Errorf("round-trip failed: got %q, want %q", redecoded, edited) |
| 319 | } |
| 320 | } |
| 321 | |
| 322 | // UTF-16 supplementary plane (surrogate pairs) |
| 323 | |
| 324 | func TestSurrogatePairRoundTrip(t *testing.T) { |
| 325 | // U+1F600 (😀) is in the supplementary plane and requires a surrogate pair. |
| 326 | original := "hello 😀 world" |
| 327 | encoded := MustEncode(original, UTF16LE) |
| 328 | decoded := string(Decode(encoded, UTF16LE)) |
| 329 | if decoded != original { |
| 330 | t.Errorf("surrogate pair round-trip failed: got %q, want %q", decoded, original) |
| 331 | } |
| 332 | } |
| 333 |