mirror of
https://github.com/navidrome/navidrome.git
synced 2026-10-08 02:17:25 +02:00
Merge b9949ce9e1 into 52135913d4
This commit is contained in:
commit
3e50b80c77
3 changed files with 168 additions and 8 deletions
|
|
@ -21,6 +21,7 @@ import (
|
|||
. "github.com/onsi/ginkgo/v2"
|
||||
. "github.com/onsi/gomega"
|
||||
"github.com/zeebo/xxh3"
|
||||
"golang.org/x/text/encoding/charmap"
|
||||
"golang.org/x/text/unicode/norm"
|
||||
)
|
||||
|
||||
|
|
@ -100,6 +101,34 @@ var _ = Describe("Playlists - Import", func() {
|
|||
Expect(pls.Tracks[0].Path).To(Equal("tests/fixtures/playlists/test.mp3"))
|
||||
})
|
||||
|
||||
It("matches accented tracks from a Windows-1252/Latin-1 encoded playlist (issue #6037)", func() {
|
||||
tmpDir := GinkgoT().TempDir()
|
||||
|
||||
// The track exists in the library with a correctly UTF-8 encoded name
|
||||
// mixing CJK folders with accented Latin characters, as reported.
|
||||
dbPath := "PokéMon Black Và White.mp3"
|
||||
|
||||
// The playlist was saved by a Windows editor as Windows-1252, so the
|
||||
// accented characters are single high bytes (é=0xE9, à=0xE0) rather
|
||||
// than UTF-8. Encode the same name to Windows-1252 for the on-disk file.
|
||||
latin1Line, err := charmap.Windows1252.NewEncoder().String(dbPath)
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
Expect([]byte(latin1Line)).To(ContainElement(byte(0xE9)))
|
||||
|
||||
plsFile := filepath.Join(tmpDir, "test.m3u")
|
||||
Expect(os.WriteFile(plsFile, []byte(latin1Line+"\n"), 0600)).To(Succeed())
|
||||
|
||||
mockLibRepo.SetData([]model.Library{{ID: 1, Path: tmpDir}})
|
||||
ds.MockedMediaFile = &mockedMediaFileFromListRepo{data: []string{dbPath}}
|
||||
ps = playlists.NewPlaylists(ds, artwork.NewUploader(ds))
|
||||
|
||||
plsFolder := &model.Folder{ID: "1", LibraryID: 1, LibraryPath: tmpDir, Path: "", Name: ""}
|
||||
pls, err := ps.ImportFromFolder(ctx, plsFolder, "test.m3u")
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
Expect(pls.Tracks).To(HaveLen(1))
|
||||
Expect(pls.Tracks[0].Path).To(Equal(dbPath))
|
||||
})
|
||||
|
||||
It("parses #EXTALBUMARTURL with HTTP URL", func() {
|
||||
conf.Server.EnableM3UExternalAlbumArt = true
|
||||
|
||||
|
|
|
|||
|
|
@ -1,26 +1,78 @@
|
|||
package ioutils
|
||||
|
||||
import (
|
||||
"bufio"
|
||||
"bytes"
|
||||
"io"
|
||||
"os"
|
||||
"unicode/utf8"
|
||||
|
||||
"golang.org/x/text/encoding/charmap"
|
||||
"golang.org/x/text/encoding/unicode"
|
||||
"golang.org/x/text/transform"
|
||||
)
|
||||
|
||||
// UTF8Reader wraps an io.Reader to handle Byte Order Mark (BOM) properly.
|
||||
// It strips UTF-8 BOM if present, and converts UTF-16 (LE/BE) to UTF-8.
|
||||
// This is particularly useful for reading user-provided text files (like LRC lyrics,
|
||||
// playlists) that may have been created on Windows, which often adds BOM markers.
|
||||
// UTF8Reader wraps an io.Reader so downstream code always sees UTF-8, whatever
|
||||
// encoding a user-provided text file (LRC lyrics, M3U playlists) happened to be
|
||||
// saved in. A Byte Order Mark is authoritative: a UTF-8 BOM is stripped and
|
||||
// UTF-16 (LE/BE) is transcoded to UTF-8, both handled as a stream.
|
||||
//
|
||||
// Without a BOM the encoding has to be judged from the whole content. The reader
|
||||
// keeps the bytes as they are when they form valid UTF-8, and otherwise decodes
|
||||
// the entire input as Windows-1252 (a superset of Latin-1). Choosing one encoding
|
||||
// for the whole file, rather than per character, matters: a genuinely
|
||||
// Windows-1252 file can contain a byte pair that happens to be valid UTF-8 (for
|
||||
// example "£" is C2 A3), and a per-character choice would decode that pair as a
|
||||
// different string than the rest of the file, so the affected path would silently
|
||||
// fail to match. This recovers accented characters from the legacy single-byte
|
||||
// encodings Windows editors still emit instead of replacing them with U+FFFD.
|
||||
//
|
||||
// Reference: https://en.wikipedia.org/wiki/Byte_order_mark
|
||||
func UTF8Reader(r io.Reader) io.Reader {
|
||||
return transform.NewReader(r, unicode.BOMOverride(unicode.UTF8.NewDecoder()))
|
||||
br := bufio.NewReader(r)
|
||||
|
||||
// A BOM is an explicit declaration, so honor it and keep the stream lazy.
|
||||
if prefix, _ := br.Peek(3); hasBOM(prefix) {
|
||||
return transform.NewReader(br, unicode.BOMOverride(unicode.UTF8.NewDecoder()))
|
||||
}
|
||||
|
||||
// No BOM: read it all so the encoding can be decided from the whole content.
|
||||
data, err := io.ReadAll(br)
|
||||
out := data
|
||||
if !utf8.Valid(data) {
|
||||
// Windows-1252 maps every byte, so this decode never fails.
|
||||
out, _ = charmap.Windows1252.NewDecoder().Bytes(data)
|
||||
}
|
||||
if err != nil {
|
||||
return io.MultiReader(bytes.NewReader(out), errorReader{err: err})
|
||||
}
|
||||
return bytes.NewReader(out)
|
||||
}
|
||||
|
||||
// UTF8ReadFile reads the named file and returns its contents as a byte slice,
|
||||
// automatically handling BOM markers. It's similar to os.ReadFile but strips
|
||||
// UTF-8 BOM and converts UTF-16 encoded files to UTF-8.
|
||||
// hasBOM reports whether b starts with a UTF-8 or UTF-16 (LE/BE) Byte Order Mark.
|
||||
func hasBOM(b []byte) bool {
|
||||
switch {
|
||||
case len(b) >= 3 && b[0] == 0xEF && b[1] == 0xBB && b[2] == 0xBF: // UTF-8
|
||||
return true
|
||||
case len(b) >= 2 && b[0] == 0xFF && b[1] == 0xFE: // UTF-16 LE
|
||||
return true
|
||||
case len(b) >= 2 && b[0] == 0xFE && b[1] == 0xFF: // UTF-16 BE
|
||||
return true
|
||||
default:
|
||||
return false
|
||||
}
|
||||
}
|
||||
|
||||
// errorReader replays a read error that surfaced while buffering the input, so
|
||||
// callers still see it lazily on their next Read.
|
||||
type errorReader struct{ err error }
|
||||
|
||||
func (e errorReader) Read([]byte) (int, error) { return 0, e.err }
|
||||
|
||||
// UTF8ReadFile reads the named file and returns its contents as UTF-8 bytes.
|
||||
// It's like os.ReadFile but runs the data through UTF8Reader, so BOMs are
|
||||
// stripped, UTF-16 is transcoded, and legacy Windows-1252/Latin-1 bytes are
|
||||
// recovered rather than replaced with U+FFFD.
|
||||
func UTF8ReadFile(filename string) ([]byte, error) {
|
||||
file, err := os.Open(filename)
|
||||
if err != nil {
|
||||
|
|
|
|||
|
|
@ -2,8 +2,10 @@ package ioutils
|
|||
|
||||
import (
|
||||
"bytes"
|
||||
"errors"
|
||||
"io"
|
||||
"testing"
|
||||
"testing/iotest"
|
||||
|
||||
. "github.com/onsi/ginkgo/v2"
|
||||
. "github.com/onsi/gomega"
|
||||
|
|
@ -81,6 +83,83 @@ var _ = Describe("UTF8Reader", func() {
|
|||
Expect(string(output)).To(Equal(""))
|
||||
})
|
||||
})
|
||||
|
||||
Context("when reading Windows-1252/Latin-1 encoded text (issue #6037)", func() {
|
||||
It("decodes accented bytes instead of emitting U+FFFD", func() {
|
||||
// "PokéMon" and "Và" with é (0xE9) and à (0xE0) as single Latin-1 bytes.
|
||||
input := []byte{'P', 'o', 'k', 0xE9, 'M', 'o', 'n', ' ', 'V', 0xE0}
|
||||
reader := UTF8Reader(bytes.NewReader(input))
|
||||
|
||||
output, err := io.ReadAll(reader)
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
Expect(string(output)).To(Equal("PokéMon Và"))
|
||||
})
|
||||
|
||||
It("decodes Windows-1252 specific bytes in the 0x80-0x9F range", func() {
|
||||
// 0x80 is the Euro sign and 0x93/0x94 are curly quotes in Windows-1252,
|
||||
// none of which exist in plain Latin-1.
|
||||
input := []byte{0x80, '5', ' ', 0x93, 'h', 'i', 0x94}
|
||||
reader := UTF8Reader(bytes.NewReader(input))
|
||||
|
||||
output, err := io.ReadAll(reader)
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
Expect(string(output)).To(Equal("€5 “hi”"))
|
||||
})
|
||||
|
||||
It("leaves valid multi-byte UTF-8 untouched", func() {
|
||||
// A path that mixes CJK and accented Latin, already valid UTF-8, must
|
||||
// pass through byte-for-byte so real UTF-8 files are never corrupted.
|
||||
input := []byte("收藏/PokéMon Và White.mp3")
|
||||
reader := UTF8Reader(bytes.NewReader(input))
|
||||
|
||||
output, err := io.ReadAll(reader)
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
Expect(output).To(Equal(input))
|
||||
})
|
||||
|
||||
It("preserves a genuine U+FFFD present in valid UTF-8", func() {
|
||||
input := []byte("a<>b")
|
||||
reader := UTF8Reader(bytes.NewReader(input))
|
||||
|
||||
output, err := io.ReadAll(reader)
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
Expect(output).To(Equal(input))
|
||||
})
|
||||
|
||||
It("decodes a byte pair that is coincidentally valid UTF-8 as Windows-1252 when the whole file is not UTF-8", func() {
|
||||
// The file is Windows-1252: "café/£.mp3" where é is 0xE9 and "£" is
|
||||
// the bytes 0xC2 0xA3. That pair alone is valid UTF-8 for "£", but the
|
||||
// standalone 0xE9 makes the file as a whole invalid UTF-8, so the whole
|
||||
// input must be read as Windows-1252 and the pair must become "£".
|
||||
input := []byte{'c', 'a', 'f', 0xE9, '/', 0xC2, 0xA3, '.', 'm', 'p', '3'}
|
||||
reader := UTF8Reader(bytes.NewReader(input))
|
||||
|
||||
output, err := io.ReadAll(reader)
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
Expect(string(output)).To(Equal("café/£.mp3"))
|
||||
})
|
||||
|
||||
It("keeps a C2 A3 pair as £ when the whole file is valid UTF-8", func() {
|
||||
// The same byte pair, in a file that is valid UTF-8 throughout, is the
|
||||
// pound sign and must be left alone.
|
||||
input := []byte("cost/£.mp3")
|
||||
reader := UTF8Reader(bytes.NewReader(input))
|
||||
|
||||
output, err := io.ReadAll(reader)
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
Expect(output).To(Equal(input))
|
||||
})
|
||||
})
|
||||
|
||||
Context("when the underlying reader fails", func() {
|
||||
It("surfaces the read error", func() {
|
||||
boom := errors.New("boom")
|
||||
reader := UTF8Reader(io.MultiReader(bytes.NewReader([]byte("abc")), iotest.ErrReader(boom)))
|
||||
|
||||
_, err := io.ReadAll(reader)
|
||||
Expect(err).To(MatchError(boom))
|
||||
})
|
||||
})
|
||||
})
|
||||
|
||||
var _ = Describe("UTF8ReadFile", func() {
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue