diff --git a/core/playlists/import_test.go b/core/playlists/import_test.go index 25960f0fe..fed2a280c 100644 --- a/core/playlists/import_test.go +++ b/core/playlists/import_test.go @@ -21,6 +21,7 @@ import ( . "github.com/onsi/ginkgo/v2" . "github.com/onsi/gomega" "github.com/zeebo/xxh3" + "golang.org/x/text/encoding/charmap" "golang.org/x/text/unicode/norm" ) @@ -100,6 +101,34 @@ var _ = Describe("Playlists - Import", func() { Expect(pls.Tracks[0].Path).To(Equal("tests/fixtures/playlists/test.mp3")) }) + It("matches accented tracks from a Windows-1252/Latin-1 encoded playlist (issue #6037)", func() { + tmpDir := GinkgoT().TempDir() + + // The track exists in the library with a correctly UTF-8 encoded name + // mixing CJK folders with accented Latin characters, as reported. + dbPath := "PokéMon Black Và White.mp3" + + // The playlist was saved by a Windows editor as Windows-1252, so the + // accented characters are single high bytes (é=0xE9, à=0xE0) rather + // than UTF-8. Encode the same name to Windows-1252 for the on-disk file. + latin1Line, err := charmap.Windows1252.NewEncoder().String(dbPath) + Expect(err).ToNot(HaveOccurred()) + Expect([]byte(latin1Line)).To(ContainElement(byte(0xE9))) + + plsFile := filepath.Join(tmpDir, "test.m3u") + Expect(os.WriteFile(plsFile, []byte(latin1Line+"\n"), 0600)).To(Succeed()) + + mockLibRepo.SetData([]model.Library{{ID: 1, Path: tmpDir}}) + ds.MockedMediaFile = &mockedMediaFileFromListRepo{data: []string{dbPath}} + ps = playlists.NewPlaylists(ds, artwork.NewUploader(ds)) + + plsFolder := &model.Folder{ID: "1", LibraryID: 1, LibraryPath: tmpDir, Path: "", Name: ""} + pls, err := ps.ImportFromFolder(ctx, plsFolder, "test.m3u") + Expect(err).ToNot(HaveOccurred()) + Expect(pls.Tracks).To(HaveLen(1)) + Expect(pls.Tracks[0].Path).To(Equal(dbPath)) + }) + It("parses #EXTALBUMARTURL with HTTP URL", func() { conf.Server.EnableM3UExternalAlbumArt = true diff --git a/utils/ioutils/ioutils.go b/utils/ioutils/ioutils.go index 89d3997f3..e33dc5861 100644 --- a/utils/ioutils/ioutils.go +++ b/utils/ioutils/ioutils.go @@ -1,26 +1,78 @@ package ioutils import ( + "bufio" + "bytes" "io" "os" + "unicode/utf8" + "golang.org/x/text/encoding/charmap" "golang.org/x/text/encoding/unicode" "golang.org/x/text/transform" ) -// UTF8Reader wraps an io.Reader to handle Byte Order Mark (BOM) properly. -// It strips UTF-8 BOM if present, and converts UTF-16 (LE/BE) to UTF-8. -// This is particularly useful for reading user-provided text files (like LRC lyrics, -// playlists) that may have been created on Windows, which often adds BOM markers. +// UTF8Reader wraps an io.Reader so downstream code always sees UTF-8, whatever +// encoding a user-provided text file (LRC lyrics, M3U playlists) happened to be +// saved in. A Byte Order Mark is authoritative: a UTF-8 BOM is stripped and +// UTF-16 (LE/BE) is transcoded to UTF-8, both handled as a stream. +// +// Without a BOM the encoding has to be judged from the whole content. The reader +// keeps the bytes as they are when they form valid UTF-8, and otherwise decodes +// the entire input as Windows-1252 (a superset of Latin-1). Choosing one encoding +// for the whole file, rather than per character, matters: a genuinely +// Windows-1252 file can contain a byte pair that happens to be valid UTF-8 (for +// example "£" is C2 A3), and a per-character choice would decode that pair as a +// different string than the rest of the file, so the affected path would silently +// fail to match. This recovers accented characters from the legacy single-byte +// encodings Windows editors still emit instead of replacing them with U+FFFD. // // Reference: https://en.wikipedia.org/wiki/Byte_order_mark func UTF8Reader(r io.Reader) io.Reader { - return transform.NewReader(r, unicode.BOMOverride(unicode.UTF8.NewDecoder())) + br := bufio.NewReader(r) + + // A BOM is an explicit declaration, so honor it and keep the stream lazy. + if prefix, _ := br.Peek(3); hasBOM(prefix) { + return transform.NewReader(br, unicode.BOMOverride(unicode.UTF8.NewDecoder())) + } + + // No BOM: read it all so the encoding can be decided from the whole content. + data, err := io.ReadAll(br) + out := data + if !utf8.Valid(data) { + // Windows-1252 maps every byte, so this decode never fails. + out, _ = charmap.Windows1252.NewDecoder().Bytes(data) + } + if err != nil { + return io.MultiReader(bytes.NewReader(out), errorReader{err: err}) + } + return bytes.NewReader(out) } -// UTF8ReadFile reads the named file and returns its contents as a byte slice, -// automatically handling BOM markers. It's similar to os.ReadFile but strips -// UTF-8 BOM and converts UTF-16 encoded files to UTF-8. +// hasBOM reports whether b starts with a UTF-8 or UTF-16 (LE/BE) Byte Order Mark. +func hasBOM(b []byte) bool { + switch { + case len(b) >= 3 && b[0] == 0xEF && b[1] == 0xBB && b[2] == 0xBF: // UTF-8 + return true + case len(b) >= 2 && b[0] == 0xFF && b[1] == 0xFE: // UTF-16 LE + return true + case len(b) >= 2 && b[0] == 0xFE && b[1] == 0xFF: // UTF-16 BE + return true + default: + return false + } +} + +// errorReader replays a read error that surfaced while buffering the input, so +// callers still see it lazily on their next Read. +type errorReader struct{ err error } + +func (e errorReader) Read([]byte) (int, error) { return 0, e.err } + +// UTF8ReadFile reads the named file and returns its contents as UTF-8 bytes. +// It's like os.ReadFile but runs the data through UTF8Reader, so BOMs are +// stripped, UTF-16 is transcoded, and legacy Windows-1252/Latin-1 bytes are +// recovered rather than replaced with U+FFFD. func UTF8ReadFile(filename string) ([]byte, error) { file, err := os.Open(filename) if err != nil { diff --git a/utils/ioutils/ioutils_test.go b/utils/ioutils/ioutils_test.go index 7f5483879..7dbdca0c2 100644 --- a/utils/ioutils/ioutils_test.go +++ b/utils/ioutils/ioutils_test.go @@ -2,8 +2,10 @@ package ioutils import ( "bytes" + "errors" "io" "testing" + "testing/iotest" . "github.com/onsi/ginkgo/v2" . "github.com/onsi/gomega" @@ -81,6 +83,83 @@ var _ = Describe("UTF8Reader", func() { Expect(string(output)).To(Equal("")) }) }) + + Context("when reading Windows-1252/Latin-1 encoded text (issue #6037)", func() { + It("decodes accented bytes instead of emitting U+FFFD", func() { + // "PokéMon" and "Và" with é (0xE9) and à (0xE0) as single Latin-1 bytes. + input := []byte{'P', 'o', 'k', 0xE9, 'M', 'o', 'n', ' ', 'V', 0xE0} + reader := UTF8Reader(bytes.NewReader(input)) + + output, err := io.ReadAll(reader) + Expect(err).ToNot(HaveOccurred()) + Expect(string(output)).To(Equal("PokéMon Và")) + }) + + It("decodes Windows-1252 specific bytes in the 0x80-0x9F range", func() { + // 0x80 is the Euro sign and 0x93/0x94 are curly quotes in Windows-1252, + // none of which exist in plain Latin-1. + input := []byte{0x80, '5', ' ', 0x93, 'h', 'i', 0x94} + reader := UTF8Reader(bytes.NewReader(input)) + + output, err := io.ReadAll(reader) + Expect(err).ToNot(HaveOccurred()) + Expect(string(output)).To(Equal("€5 “hi”")) + }) + + It("leaves valid multi-byte UTF-8 untouched", func() { + // A path that mixes CJK and accented Latin, already valid UTF-8, must + // pass through byte-for-byte so real UTF-8 files are never corrupted. + input := []byte("收藏/PokéMon Và White.mp3") + reader := UTF8Reader(bytes.NewReader(input)) + + output, err := io.ReadAll(reader) + Expect(err).ToNot(HaveOccurred()) + Expect(output).To(Equal(input)) + }) + + It("preserves a genuine U+FFFD present in valid UTF-8", func() { + input := []byte("a�b") + reader := UTF8Reader(bytes.NewReader(input)) + + output, err := io.ReadAll(reader) + Expect(err).ToNot(HaveOccurred()) + Expect(output).To(Equal(input)) + }) + + It("decodes a byte pair that is coincidentally valid UTF-8 as Windows-1252 when the whole file is not UTF-8", func() { + // The file is Windows-1252: "café/£.mp3" where é is 0xE9 and "£" is + // the bytes 0xC2 0xA3. That pair alone is valid UTF-8 for "£", but the + // standalone 0xE9 makes the file as a whole invalid UTF-8, so the whole + // input must be read as Windows-1252 and the pair must become "£". + input := []byte{'c', 'a', 'f', 0xE9, '/', 0xC2, 0xA3, '.', 'm', 'p', '3'} + reader := UTF8Reader(bytes.NewReader(input)) + + output, err := io.ReadAll(reader) + Expect(err).ToNot(HaveOccurred()) + Expect(string(output)).To(Equal("café/£.mp3")) + }) + + It("keeps a C2 A3 pair as £ when the whole file is valid UTF-8", func() { + // The same byte pair, in a file that is valid UTF-8 throughout, is the + // pound sign and must be left alone. + input := []byte("cost/£.mp3") + reader := UTF8Reader(bytes.NewReader(input)) + + output, err := io.ReadAll(reader) + Expect(err).ToNot(HaveOccurred()) + Expect(output).To(Equal(input)) + }) + }) + + Context("when the underlying reader fails", func() { + It("surfaces the read error", func() { + boom := errors.New("boom") + reader := UTF8Reader(io.MultiReader(bytes.NewReader([]byte("abc")), iotest.ErrReader(boom))) + + _, err := io.ReadAll(reader) + Expect(err).To(MatchError(boom)) + }) + }) }) var _ = Describe("UTF8ReadFile", func() {