This commit is contained in:
Steve Jason 2026-10-08 01:55:22 +08:00 • committed by GitHub
commit 3e50b80c77
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
3 changed files with 168 additions and 8 deletions

View file

@ -21,6 +21,7 @@ import (
. "github.com/onsi/ginkgo/v2"
. "github.com/onsi/gomega"
"github.com/zeebo/xxh3"
"golang.org/x/text/encoding/charmap"
"golang.org/x/text/unicode/norm"
)
@ -100,6 +101,34 @@ var _ = Describe("Playlists - Import", func() {
Expect(pls.Tracks[0].Path).To(Equal("tests/fixtures/playlists/test.mp3"))
})
It("matches accented tracks from a Windows-1252/Latin-1 encoded playlist (issue #6037)", func() {
tmpDir := GinkgoT().TempDir()
// The track exists in the library with a correctly UTF-8 encoded name
// mixing CJK folders with accented Latin characters, as reported.
dbPath := "PokéMon Black Và White.mp3"
// The playlist was saved by a Windows editor as Windows-1252, so the
// accented characters are single high bytes (é=0xE9, à=0xE0) rather
// than UTF-8. Encode the same name to Windows-1252 for the on-disk file.
latin1Line, err := charmap.Windows1252.NewEncoder().String(dbPath)
Expect(err).ToNot(HaveOccurred())
Expect([]byte(latin1Line)).To(ContainElement(byte(0xE9)))
plsFile := filepath.Join(tmpDir, "test.m3u")
Expect(os.WriteFile(plsFile, []byte(latin1Line+"\n"), 0600)).To(Succeed())
mockLibRepo.SetData([]model.Library{{ID: 1, Path: tmpDir}})
ds.MockedMediaFile = &mockedMediaFileFromListRepo{data: []string{dbPath}}
ps = playlists.NewPlaylists(ds, artwork.NewUploader(ds))
plsFolder := &model.Folder{ID: "1", LibraryID: 1, LibraryPath: tmpDir, Path: "", Name: ""}
pls, err := ps.ImportFromFolder(ctx, plsFolder, "test.m3u")
Expect(err).ToNot(HaveOccurred())
Expect(pls.Tracks).To(HaveLen(1))
Expect(pls.Tracks[0].Path).To(Equal(dbPath))
})
It("parses #EXTALBUMARTURL with HTTP URL", func() {
conf.Server.EnableM3UExternalAlbumArt = true

View file

@ -1,26 +1,78 @@
package ioutils
import (
"bufio"
"bytes"
"io"
"os"
"unicode/utf8"
"golang.org/x/text/encoding/charmap"
"golang.org/x/text/encoding/unicode"
"golang.org/x/text/transform"
)
// UTF8Reader wraps an io.Reader to handle Byte Order Mark (BOM) properly.
// It strips UTF-8 BOM if present, and converts UTF-16 (LE/BE) to UTF-8.
// This is particularly useful for reading user-provided text files (like LRC lyrics,
// playlists) that may have been created on Windows, which often adds BOM markers.
// UTF8Reader wraps an io.Reader so downstream code always sees UTF-8, whatever
// encoding a user-provided text file (LRC lyrics, M3U playlists) happened to be
// saved in. A Byte Order Mark is authoritative: a UTF-8 BOM is stripped and
// UTF-16 (LE/BE) is transcoded to UTF-8, both handled as a stream.
//
// Without a BOM the encoding has to be judged from the whole content. The reader
// keeps the bytes as they are when they form valid UTF-8, and otherwise decodes
// the entire input as Windows-1252 (a superset of Latin-1). Choosing one encoding
// for the whole file, rather than per character, matters: a genuinely
// Windows-1252 file can contain a byte pair that happens to be valid UTF-8 (for
// example "£" is C2 A3), and a per-character choice would decode that pair as a
// different string than the rest of the file, so the affected path would silently
// fail to match. This recovers accented characters from the legacy single-byte
// encodings Windows editors still emit instead of replacing them with U+FFFD.
//
// Reference: https://en.wikipedia.org/wiki/Byte_order_mark
func UTF8Reader(r io.Reader) io.Reader {
return transform.NewReader(r, unicode.BOMOverride(unicode.UTF8.NewDecoder()))
br := bufio.NewReader(r)
// A BOM is an explicit declaration, so honor it and keep the stream lazy.
if prefix, _ := br.Peek(3); hasBOM(prefix) {
return transform.NewReader(br, unicode.BOMOverride(unicode.UTF8.NewDecoder()))
}
// No BOM: read it all so the encoding can be decided from the whole content.
data, err := io.ReadAll(br)
out := data
if !utf8.Valid(data) {
// Windows-1252 maps every byte, so this decode never fails.
out, _ = charmap.Windows1252.NewDecoder().Bytes(data)
}
if err != nil {
return io.MultiReader(bytes.NewReader(out), errorReader{err: err})
}
return bytes.NewReader(out)
}
// UTF8ReadFile reads the named file and returns its contents as a byte slice,
// automatically handling BOM markers. It's similar to os.ReadFile but strips
// UTF-8 BOM and converts UTF-16 encoded files to UTF-8.
// hasBOM reports whether b starts with a UTF-8 or UTF-16 (LE/BE) Byte Order Mark.
func hasBOM(b []byte) bool {
switch {
case len(b) >= 3 && b[0] == 0xEF && b[1] == 0xBB && b[2] == 0xBF: // UTF-8
return true
case len(b) >= 2 && b[0] == 0xFF && b[1] == 0xFE: // UTF-16 LE
return true
case len(b) >= 2 && b[0] == 0xFE && b[1] == 0xFF: // UTF-16 BE
return true
default:
return false
}
}
// errorReader replays a read error that surfaced while buffering the input, so
// callers still see it lazily on their next Read.
type errorReader struct{ err error }
func (e errorReader) Read([]byte) (int, error) { return 0, e.err }
// UTF8ReadFile reads the named file and returns its contents as UTF-8 bytes.
// It's like os.ReadFile but runs the data through UTF8Reader, so BOMs are
// stripped, UTF-16 is transcoded, and legacy Windows-1252/Latin-1 bytes are
// recovered rather than replaced with U+FFFD.
func UTF8ReadFile(filename string) ([]byte, error) {
file, err := os.Open(filename)
if err != nil {

View file

@ -2,8 +2,10 @@ package ioutils
import (
"bytes"
"errors"
"io"
"testing"
"testing/iotest"
. "github.com/onsi/ginkgo/v2"
. "github.com/onsi/gomega"
@ -81,6 +83,83 @@ var _ = Describe("UTF8Reader", func() {
Expect(string(output)).To(Equal(""))
})
})
Context("when reading Windows-1252/Latin-1 encoded text (issue #6037)", func() {
It("decodes accented bytes instead of emitting U+FFFD", func() {
// "PokéMon" and "Và" with é (0xE9) and à (0xE0) as single Latin-1 bytes.
input := []byte{'P', 'o', 'k', 0xE9, 'M', 'o', 'n', ' ', 'V', 0xE0}
reader := UTF8Reader(bytes.NewReader(input))
output, err := io.ReadAll(reader)
Expect(err).ToNot(HaveOccurred())
Expect(string(output)).To(Equal("PokéMon Và"))
})
It("decodes Windows-1252 specific bytes in the 0x80-0x9F range", func() {
// 0x80 is the Euro sign and 0x93/0x94 are curly quotes in Windows-1252,
// none of which exist in plain Latin-1.
input := []byte{0x80, '5', ' ', 0x93, 'h', 'i', 0x94}
reader := UTF8Reader(bytes.NewReader(input))
output, err := io.ReadAll(reader)
Expect(err).ToNot(HaveOccurred())
Expect(string(output)).To(Equal("€5 “hi”"))
})
It("leaves valid multi-byte UTF-8 untouched", func() {
// A path that mixes CJK and accented Latin, already valid UTF-8, must
// pass through byte-for-byte so real UTF-8 files are never corrupted.
input := []byte("收藏/PokéMon Và White.mp3")
reader := UTF8Reader(bytes.NewReader(input))
output, err := io.ReadAll(reader)
Expect(err).ToNot(HaveOccurred())
Expect(output).To(Equal(input))
})
It("preserves a genuine U+FFFD present in valid UTF-8", func() {
input := []byte("a<>b")
reader := UTF8Reader(bytes.NewReader(input))
output, err := io.ReadAll(reader)
Expect(err).ToNot(HaveOccurred())
Expect(output).To(Equal(input))
})
It("decodes a byte pair that is coincidentally valid UTF-8 as Windows-1252 when the whole file is not UTF-8", func() {
// The file is Windows-1252: "café/£.mp3" where é is 0xE9 and "£" is
// the bytes 0xC2 0xA3. That pair alone is valid UTF-8 for "£", but the
// standalone 0xE9 makes the file as a whole invalid UTF-8, so the whole
// input must be read as Windows-1252 and the pair must become "£".
input := []byte{'c', 'a', 'f', 0xE9, '/', 0xC2, 0xA3, '.', 'm', 'p', '3'}
reader := UTF8Reader(bytes.NewReader(input))
output, err := io.ReadAll(reader)
Expect(err).ToNot(HaveOccurred())
Expect(string(output)).To(Equal("café/£.mp3"))
})
It("keeps a C2 A3 pair as £ when the whole file is valid UTF-8", func() {
// The same byte pair, in a file that is valid UTF-8 throughout, is the
// pound sign and must be left alone.
input := []byte("cost/£.mp3")
reader := UTF8Reader(bytes.NewReader(input))
output, err := io.ReadAll(reader)
Expect(err).ToNot(HaveOccurred())
Expect(output).To(Equal(input))
})
})
Context("when the underlying reader fails", func() {
It("surfaces the read error", func() {
boom := errors.New("boom")
reader := UTF8Reader(io.MultiReader(bytes.NewReader([]byte("abc")), iotest.ErrReader(boom)))
_, err := io.ReadAll(reader)
Expect(err).To(MatchError(boom))
})
})
})
var _ = Describe("UTF8ReadFile", func() {