From da0899de2b66282b294e958e01e3baf0525bf422 Mon Sep 17 00:00:00 2001 From: Steve Date: Tue, 1 Sep 2026 12:48:56 +0530 Subject: [PATCH 1/3] fix(ioutils): recover Latin-1/Windows-1252 text in playlists and lyrics - #6037 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit UTF8Reader decoded non-BOM input strictly as UTF-8, so a byte that was not valid UTF-8 became the replacement character U+FFFD. Playlists and lyrics saved by Windows editors are commonly Windows-1252/Latin-1, where accented characters are single high bytes (for example é as 0xE9). Those bytes were mangled into U+FFFD, so an M3U entry like 'PokéMon' no longer matched the scanned library path and the track was silently skipped, while pure-ASCII entries in the same playlist imported fine. UTF8Reader now keeps valid UTF-8 untouched and decodes any invalid byte as Windows-1252 (a superset of Latin-1), recovering the accented characters instead of losing them. BOM handling and UTF-16 transcoding are unchanged. Closes #6037 Signed-off-by: Steve --- core/playlists/import_test.go | 29 ++++++++++++++ utils/ioutils/ioutils.go | 75 +++++++++++++++++++++++++++++++---- utils/ioutils/ioutils_test.go | 43 ++++++++++++++++++++ 3 files changed, 139 insertions(+), 8 deletions(-) diff --git a/core/playlists/import_test.go b/core/playlists/import_test.go index 445561266..2fa592998 100644 --- a/core/playlists/import_test.go +++ b/core/playlists/import_test.go @@ -21,6 +21,7 @@ import ( . "github.com/onsi/ginkgo/v2" . "github.com/onsi/gomega" "github.com/zeebo/xxh3" + "golang.org/x/text/encoding/charmap" "golang.org/x/text/unicode/norm" ) @@ -100,6 +101,34 @@ var _ = Describe("Playlists - Import", func() { Expect(pls.Tracks[0].Path).To(Equal("tests/fixtures/playlists/test.mp3")) }) + It("matches accented tracks from a Windows-1252/Latin-1 encoded playlist (issue #6037)", func() { + tmpDir := GinkgoT().TempDir() + + // The track exists in the library with a correctly UTF-8 encoded name + // mixing CJK folders with accented Latin characters, as reported. + dbPath := "PokéMon Black Và White.mp3" + + // The playlist was saved by a Windows editor as Windows-1252, so the + // accented characters are single high bytes (é=0xE9, à=0xE0) rather + // than UTF-8. Encode the same name to Windows-1252 for the on-disk file. + latin1Line, err := charmap.Windows1252.NewEncoder().String(dbPath) + Expect(err).ToNot(HaveOccurred()) + Expect([]byte(latin1Line)).To(ContainElement(byte(0xE9))) + + plsFile := filepath.Join(tmpDir, "test.m3u") + Expect(os.WriteFile(plsFile, []byte(latin1Line+"\n"), 0600)).To(Succeed()) + + mockLibRepo.SetData([]model.Library{{ID: 1, Path: tmpDir}}) + ds.MockedMediaFile = &mockedMediaFileFromListRepo{data: []string{dbPath}} + ps = playlists.NewPlaylists(ds, artwork.NewUploader(ds)) + + plsFolder := &model.Folder{ID: "1", LibraryID: 1, LibraryPath: tmpDir, Path: "", Name: ""} + pls, err := ps.ImportFromFolder(ctx, plsFolder, "test.m3u") + Expect(err).ToNot(HaveOccurred()) + Expect(pls.Tracks).To(HaveLen(1)) + Expect(pls.Tracks[0].Path).To(Equal(dbPath)) + }) + It("parses #EXTALBUMARTURL with HTTP URL", func() { conf.Server.EnableM3UExternalAlbumArt = true diff --git a/utils/ioutils/ioutils.go b/utils/ioutils/ioutils.go index 89d3997f3..2e4fa2ad7 100644 --- a/utils/ioutils/ioutils.go +++ b/utils/ioutils/ioutils.go @@ -3,24 +3,83 @@ package ioutils import ( "io" "os" + "unicode/utf8" + "golang.org/x/text/encoding/charmap" "golang.org/x/text/encoding/unicode" "golang.org/x/text/transform" ) -// UTF8Reader wraps an io.Reader to handle Byte Order Mark (BOM) properly. -// It strips UTF-8 BOM if present, and converts UTF-16 (LE/BE) to UTF-8. -// This is particularly useful for reading user-provided text files (like LRC lyrics, -// playlists) that may have been created on Windows, which often adds BOM markers. +// UTF8Reader wraps an io.Reader so downstream code always sees UTF-8, whatever +// encoding a user-provided text file (LRC lyrics, M3U playlists) happened to be +// saved in. It honors a Byte Order Mark when present: a UTF-8 BOM is stripped, +// and UTF-16 (LE/BE) is transcoded to UTF-8. When there is no BOM the bytes are +// read as UTF-8, and any byte that is not valid UTF-8 is decoded as Windows-1252 +// (a superset of Latin-1). That recovers accented characters from the legacy +// single-byte encodings Windows editors still emit, instead of replacing them +// with U+FFFD and losing the match. Valid UTF-8 always passes through untouched. // // Reference: https://en.wikipedia.org/wiki/Byte_order_mark func UTF8Reader(r io.Reader) io.Reader { - return transform.NewReader(r, unicode.BOMOverride(unicode.UTF8.NewDecoder())) + return transform.NewReader(r, unicode.BOMOverride(utf8OrWindows1252{})) } -// UTF8ReadFile reads the named file and returns its contents as a byte slice, -// automatically handling BOM markers. It's similar to os.ReadFile but strips -// UTF-8 BOM and converts UTF-16 encoded files to UTF-8. +// utf8OrWindows1252 passes valid UTF-8 through unchanged and decodes any invalid +// byte as Windows-1252. UTF-8 is preferred, so well-formed input is never +// altered; the fallback only rescues bytes that could not be valid UTF-8, which +// is the hallmark of a legacy Latin-1/Windows-1252 file. +type utf8OrWindows1252 struct{} + +func (utf8OrWindows1252) Reset() {} + +func (utf8OrWindows1252) Transform(dst, src []byte, atEOF bool) (nDst, nSrc int, err error) { + for nSrc < len(src) { + b := src[nSrc] + + // ASCII fast path: identical in UTF-8 and Windows-1252. + if b < utf8.RuneSelf { + if nDst >= len(dst) { + return nDst, nSrc, transform.ErrShortDst + } + dst[nDst] = b + nDst++ + nSrc++ + continue + } + + // A byte >= 0x80 begins a multi-byte UTF-8 sequence. If the remaining + // input might still hold an incomplete sequence, ask for more before + // judging it invalid, so a rune split across reads is not misdecoded. + if !atEOF && !utf8.FullRune(src[nSrc:]) { + return nDst, nSrc, transform.ErrShortSrc + } + + if r, size := utf8.DecodeRune(src[nSrc:]); r != utf8.RuneError || size > 1 { + // Valid UTF-8 rune (including a genuine U+FFFD): copy it verbatim. + if nDst+size > len(dst) { + return nDst, nSrc, transform.ErrShortDst + } + copy(dst[nDst:], src[nSrc:nSrc+size]) + nDst += size + nSrc += size + continue + } + + // Not valid UTF-8: decode this single byte as Windows-1252. + r := charmap.Windows1252.DecodeByte(b) + if nDst+utf8.RuneLen(r) > len(dst) { + return nDst, nSrc, transform.ErrShortDst + } + nDst += utf8.EncodeRune(dst[nDst:], r) + nSrc++ + } + return nDst, nSrc, nil +} + +// UTF8ReadFile reads the named file and returns its contents as UTF-8 bytes. +// It's like os.ReadFile but runs the data through UTF8Reader, so BOMs are +// stripped, UTF-16 is transcoded, and legacy Windows-1252/Latin-1 bytes are +// recovered rather than replaced with U+FFFD. func UTF8ReadFile(filename string) ([]byte, error) { file, err := os.Open(filename) if err != nil { diff --git a/utils/ioutils/ioutils_test.go b/utils/ioutils/ioutils_test.go index 7f5483879..d1c188865 100644 --- a/utils/ioutils/ioutils_test.go +++ b/utils/ioutils/ioutils_test.go @@ -81,6 +81,49 @@ var _ = Describe("UTF8Reader", func() { Expect(string(output)).To(Equal("")) }) }) + + Context("when reading Windows-1252/Latin-1 encoded text (issue #6037)", func() { + It("decodes accented bytes instead of emitting U+FFFD", func() { + // "PokéMon" and "Và" with é (0xE9) and à (0xE0) as single Latin-1 bytes. + input := []byte{'P', 'o', 'k', 0xE9, 'M', 'o', 'n', ' ', 'V', 0xE0} + reader := UTF8Reader(bytes.NewReader(input)) + + output, err := io.ReadAll(reader) + Expect(err).ToNot(HaveOccurred()) + Expect(string(output)).To(Equal("PokéMon Và")) + }) + + It("decodes Windows-1252 specific bytes in the 0x80-0x9F range", func() { + // 0x80 is the Euro sign and 0x93/0x94 are curly quotes in Windows-1252, + // none of which exist in plain Latin-1. + input := []byte{0x80, '5', ' ', 0x93, 'h', 'i', 0x94} + reader := UTF8Reader(bytes.NewReader(input)) + + output, err := io.ReadAll(reader) + Expect(err).ToNot(HaveOccurred()) + Expect(string(output)).To(Equal("€5 “hi”")) + }) + + It("leaves valid multi-byte UTF-8 untouched", func() { + // A path that mixes CJK and accented Latin, already valid UTF-8, must + // pass through byte-for-byte so real UTF-8 files are never corrupted. + input := []byte("收藏/PokéMon Và White.mp3") + reader := UTF8Reader(bytes.NewReader(input)) + + output, err := io.ReadAll(reader) + Expect(err).ToNot(HaveOccurred()) + Expect(output).To(Equal(input)) + }) + + It("preserves a genuine U+FFFD present in valid UTF-8", func() { + input := []byte("a�b") + reader := UTF8Reader(bytes.NewReader(input)) + + output, err := io.ReadAll(reader) + Expect(err).ToNot(HaveOccurred()) + Expect(output).To(Equal(input)) + }) + }) }) var _ = Describe("UTF8ReadFile", func() { From c5b0b60687653cc244ae55f511f00dc3fc82cde0 Mon Sep 17 00:00:00 2001 From: Steve Date: Tue, 1 Sep 2026 17:32:24 +0530 Subject: [PATCH 2/3] test(ioutils): cover utf8OrWindows1252 short-buffer and edge branches - #6037 Drive the transformer directly to exercise the ErrShortDst guards (ASCII, valid multi-byte, and Windows-1252 paths), the ErrShortSrc wait for an incomplete trailing sequence, Reset, and the undefined-byte to U+FFFD case. Brings utils/ioutils/ioutils.go back to full statement coverage. Signed-off-by: Steve --- utils/ioutils/ioutils_test.go | 55 +++++++++++++++++++++++++++++++++++ 1 file changed, 55 insertions(+) diff --git a/utils/ioutils/ioutils_test.go b/utils/ioutils/ioutils_test.go index d1c188865..15663971e 100644 --- a/utils/ioutils/ioutils_test.go +++ b/utils/ioutils/ioutils_test.go @@ -7,6 +7,7 @@ import ( . "github.com/onsi/ginkgo/v2" . "github.com/onsi/gomega" + "golang.org/x/text/transform" ) func TestIOUtils(t *testing.T) { @@ -126,6 +127,60 @@ var _ = Describe("UTF8Reader", func() { }) }) +var _ = Describe("utf8OrWindows1252 transformer", func() { + var t utf8OrWindows1252 + + It("has a no-op Reset", func() { + Expect(t.Reset).ToNot(Panic()) + }) + + It("returns ErrShortDst when an ASCII byte does not fit the destination", func() { + nDst, nSrc, err := t.Transform(make([]byte, 0), []byte("a"), true) + + Expect(err).To(MatchError(transform.ErrShortDst)) + Expect(nDst).To(Equal(0)) + Expect(nSrc).To(Equal(0)) + }) + + It("returns ErrShortDst when a valid multi-byte UTF-8 rune does not fit", func() { + // "é" is 0xC3 0xA9 in UTF-8 and needs a 2-byte destination. + nDst, nSrc, err := t.Transform(make([]byte, 1), []byte("é"), true) + + Expect(err).To(MatchError(transform.ErrShortDst)) + Expect(nDst).To(Equal(0)) + Expect(nSrc).To(Equal(0)) + }) + + It("returns ErrShortDst when a Windows-1252 decoded rune does not fit", func() { + // 0xE9 decodes to "é", whose UTF-8 form needs 2 bytes. + nDst, nSrc, err := t.Transform(make([]byte, 1), []byte{0xE9}, true) + + Expect(err).To(MatchError(transform.ErrShortDst)) + Expect(nDst).To(Equal(0)) + Expect(nSrc).To(Equal(0)) + }) + + It("returns ErrShortSrc for an incomplete trailing sequence when more may follow", func() { + // 0xC3 starts a 2-byte sequence; with no second byte and not at EOF, the + // decoder must wait for more input rather than decode it as Latin-1. + nDst, nSrc, err := t.Transform(make([]byte, 8), []byte{0xC3}, false) + + Expect(err).To(MatchError(transform.ErrShortSrc)) + Expect(nDst).To(Equal(0)) + Expect(nSrc).To(Equal(0)) + }) + + It("maps an undefined Windows-1252 byte to U+FFFD at EOF", func() { + // 0x81 is undefined in Windows-1252, so it decodes to the replacement char. + dst := make([]byte, 8) + nDst, nSrc, err := t.Transform(dst, []byte{0x81}, true) + + Expect(err).ToNot(HaveOccurred()) + Expect(nSrc).To(Equal(1)) + Expect(string(dst[:nDst])).To(Equal("�")) + }) +}) + var _ = Describe("UTF8ReadFile", func() { Context("when reading a file with UTF-8 BOM", func() { It("strips the BOM marker", func() { From b9949ce9e154d0d698e8dde8af70e3401af0805f Mon Sep 17 00:00:00 2001 From: Steve Date: Thu, 3 Sep 2026 10:20:16 +0530 Subject: [PATCH 3/3] fix(ioutils): decide Latin-1 fallback per file, not per character - #6037 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Addresses the Codex review on #6063. The previous version chose the encoding one character at a time, so a genuinely Windows-1252 file that happened to contain a byte pair which is also valid UTF-8 (for example '£' as C2 A3) had that pair decoded differently from the rest of the file, leaving the affected track unmatched. UTF8Reader now judges the whole BOM-less input: it keeps the bytes when they are valid UTF-8 and otherwise decodes the entire input as Windows-1252. BOM and UTF-16 handling stay as a stream and are unchanged. Signed-off-by: Steve --- utils/ioutils/ioutils.go | 107 ++++++++++++++++------------------ utils/ioutils/ioutils_test.go | 81 ++++++++++--------------- 2 files changed, 81 insertions(+), 107 deletions(-) diff --git a/utils/ioutils/ioutils.go b/utils/ioutils/ioutils.go index 2e4fa2ad7..e33dc5861 100644 --- a/utils/ioutils/ioutils.go +++ b/utils/ioutils/ioutils.go @@ -1,6 +1,8 @@ package ioutils import ( + "bufio" + "bytes" "io" "os" "unicode/utf8" @@ -12,70 +14,61 @@ import ( // UTF8Reader wraps an io.Reader so downstream code always sees UTF-8, whatever // encoding a user-provided text file (LRC lyrics, M3U playlists) happened to be -// saved in. It honors a Byte Order Mark when present: a UTF-8 BOM is stripped, -// and UTF-16 (LE/BE) is transcoded to UTF-8. When there is no BOM the bytes are -// read as UTF-8, and any byte that is not valid UTF-8 is decoded as Windows-1252 -// (a superset of Latin-1). That recovers accented characters from the legacy -// single-byte encodings Windows editors still emit, instead of replacing them -// with U+FFFD and losing the match. Valid UTF-8 always passes through untouched. +// saved in. A Byte Order Mark is authoritative: a UTF-8 BOM is stripped and +// UTF-16 (LE/BE) is transcoded to UTF-8, both handled as a stream. +// +// Without a BOM the encoding has to be judged from the whole content. The reader +// keeps the bytes as they are when they form valid UTF-8, and otherwise decodes +// the entire input as Windows-1252 (a superset of Latin-1). Choosing one encoding +// for the whole file, rather than per character, matters: a genuinely +// Windows-1252 file can contain a byte pair that happens to be valid UTF-8 (for +// example "£" is C2 A3), and a per-character choice would decode that pair as a +// different string than the rest of the file, so the affected path would silently +// fail to match. This recovers accented characters from the legacy single-byte +// encodings Windows editors still emit instead of replacing them with U+FFFD. // // Reference: https://en.wikipedia.org/wiki/Byte_order_mark func UTF8Reader(r io.Reader) io.Reader { - return transform.NewReader(r, unicode.BOMOverride(utf8OrWindows1252{})) -} + br := bufio.NewReader(r) -// utf8OrWindows1252 passes valid UTF-8 through unchanged and decodes any invalid -// byte as Windows-1252. UTF-8 is preferred, so well-formed input is never -// altered; the fallback only rescues bytes that could not be valid UTF-8, which -// is the hallmark of a legacy Latin-1/Windows-1252 file. -type utf8OrWindows1252 struct{} - -func (utf8OrWindows1252) Reset() {} - -func (utf8OrWindows1252) Transform(dst, src []byte, atEOF bool) (nDst, nSrc int, err error) { - for nSrc < len(src) { - b := src[nSrc] - - // ASCII fast path: identical in UTF-8 and Windows-1252. - if b < utf8.RuneSelf { - if nDst >= len(dst) { - return nDst, nSrc, transform.ErrShortDst - } - dst[nDst] = b - nDst++ - nSrc++ - continue - } - - // A byte >= 0x80 begins a multi-byte UTF-8 sequence. If the remaining - // input might still hold an incomplete sequence, ask for more before - // judging it invalid, so a rune split across reads is not misdecoded. - if !atEOF && !utf8.FullRune(src[nSrc:]) { - return nDst, nSrc, transform.ErrShortSrc - } - - if r, size := utf8.DecodeRune(src[nSrc:]); r != utf8.RuneError || size > 1 { - // Valid UTF-8 rune (including a genuine U+FFFD): copy it verbatim. - if nDst+size > len(dst) { - return nDst, nSrc, transform.ErrShortDst - } - copy(dst[nDst:], src[nSrc:nSrc+size]) - nDst += size - nSrc += size - continue - } - - // Not valid UTF-8: decode this single byte as Windows-1252. - r := charmap.Windows1252.DecodeByte(b) - if nDst+utf8.RuneLen(r) > len(dst) { - return nDst, nSrc, transform.ErrShortDst - } - nDst += utf8.EncodeRune(dst[nDst:], r) - nSrc++ + // A BOM is an explicit declaration, so honor it and keep the stream lazy. + if prefix, _ := br.Peek(3); hasBOM(prefix) { + return transform.NewReader(br, unicode.BOMOverride(unicode.UTF8.NewDecoder())) } - return nDst, nSrc, nil + + // No BOM: read it all so the encoding can be decided from the whole content. + data, err := io.ReadAll(br) + out := data + if !utf8.Valid(data) { + // Windows-1252 maps every byte, so this decode never fails. + out, _ = charmap.Windows1252.NewDecoder().Bytes(data) + } + if err != nil { + return io.MultiReader(bytes.NewReader(out), errorReader{err: err}) + } + return bytes.NewReader(out) } +// hasBOM reports whether b starts with a UTF-8 or UTF-16 (LE/BE) Byte Order Mark. +func hasBOM(b []byte) bool { + switch { + case len(b) >= 3 && b[0] == 0xEF && b[1] == 0xBB && b[2] == 0xBF: // UTF-8 + return true + case len(b) >= 2 && b[0] == 0xFF && b[1] == 0xFE: // UTF-16 LE + return true + case len(b) >= 2 && b[0] == 0xFE && b[1] == 0xFF: // UTF-16 BE + return true + default: + return false + } +} + +// errorReader replays a read error that surfaced while buffering the input, so +// callers still see it lazily on their next Read. +type errorReader struct{ err error } + +func (e errorReader) Read([]byte) (int, error) { return 0, e.err } + // UTF8ReadFile reads the named file and returns its contents as UTF-8 bytes. // It's like os.ReadFile but runs the data through UTF8Reader, so BOMs are // stripped, UTF-16 is transcoded, and legacy Windows-1252/Latin-1 bytes are diff --git a/utils/ioutils/ioutils_test.go b/utils/ioutils/ioutils_test.go index 15663971e..7dbdca0c2 100644 --- a/utils/ioutils/ioutils_test.go +++ b/utils/ioutils/ioutils_test.go @@ -2,12 +2,13 @@ package ioutils import ( "bytes" + "errors" "io" "testing" + "testing/iotest" . "github.com/onsi/ginkgo/v2" . "github.com/onsi/gomega" - "golang.org/x/text/transform" ) func TestIOUtils(t *testing.T) { @@ -124,60 +125,40 @@ var _ = Describe("UTF8Reader", func() { Expect(err).ToNot(HaveOccurred()) Expect(output).To(Equal(input)) }) - }) -}) -var _ = Describe("utf8OrWindows1252 transformer", func() { - var t utf8OrWindows1252 + It("decodes a byte pair that is coincidentally valid UTF-8 as Windows-1252 when the whole file is not UTF-8", func() { + // The file is Windows-1252: "café/£.mp3" where é is 0xE9 and "£" is + // the bytes 0xC2 0xA3. That pair alone is valid UTF-8 for "£", but the + // standalone 0xE9 makes the file as a whole invalid UTF-8, so the whole + // input must be read as Windows-1252 and the pair must become "£". + input := []byte{'c', 'a', 'f', 0xE9, '/', 0xC2, 0xA3, '.', 'm', 'p', '3'} + reader := UTF8Reader(bytes.NewReader(input)) - It("has a no-op Reset", func() { - Expect(t.Reset).ToNot(Panic()) + output, err := io.ReadAll(reader) + Expect(err).ToNot(HaveOccurred()) + Expect(string(output)).To(Equal("café/£.mp3")) + }) + + It("keeps a C2 A3 pair as £ when the whole file is valid UTF-8", func() { + // The same byte pair, in a file that is valid UTF-8 throughout, is the + // pound sign and must be left alone. + input := []byte("cost/£.mp3") + reader := UTF8Reader(bytes.NewReader(input)) + + output, err := io.ReadAll(reader) + Expect(err).ToNot(HaveOccurred()) + Expect(output).To(Equal(input)) + }) }) - It("returns ErrShortDst when an ASCII byte does not fit the destination", func() { - nDst, nSrc, err := t.Transform(make([]byte, 0), []byte("a"), true) + Context("when the underlying reader fails", func() { + It("surfaces the read error", func() { + boom := errors.New("boom") + reader := UTF8Reader(io.MultiReader(bytes.NewReader([]byte("abc")), iotest.ErrReader(boom))) - Expect(err).To(MatchError(transform.ErrShortDst)) - Expect(nDst).To(Equal(0)) - Expect(nSrc).To(Equal(0)) - }) - - It("returns ErrShortDst when a valid multi-byte UTF-8 rune does not fit", func() { - // "é" is 0xC3 0xA9 in UTF-8 and needs a 2-byte destination. - nDst, nSrc, err := t.Transform(make([]byte, 1), []byte("é"), true) - - Expect(err).To(MatchError(transform.ErrShortDst)) - Expect(nDst).To(Equal(0)) - Expect(nSrc).To(Equal(0)) - }) - - It("returns ErrShortDst when a Windows-1252 decoded rune does not fit", func() { - // 0xE9 decodes to "é", whose UTF-8 form needs 2 bytes. - nDst, nSrc, err := t.Transform(make([]byte, 1), []byte{0xE9}, true) - - Expect(err).To(MatchError(transform.ErrShortDst)) - Expect(nDst).To(Equal(0)) - Expect(nSrc).To(Equal(0)) - }) - - It("returns ErrShortSrc for an incomplete trailing sequence when more may follow", func() { - // 0xC3 starts a 2-byte sequence; with no second byte and not at EOF, the - // decoder must wait for more input rather than decode it as Latin-1. - nDst, nSrc, err := t.Transform(make([]byte, 8), []byte{0xC3}, false) - - Expect(err).To(MatchError(transform.ErrShortSrc)) - Expect(nDst).To(Equal(0)) - Expect(nSrc).To(Equal(0)) - }) - - It("maps an undefined Windows-1252 byte to U+FFFD at EOF", func() { - // 0x81 is undefined in Windows-1252, so it decodes to the replacement char. - dst := make([]byte, 8) - nDst, nSrc, err := t.Transform(dst, []byte{0x81}, true) - - Expect(err).ToNot(HaveOccurred()) - Expect(nSrc).To(Equal(1)) - Expect(string(dst[:nDst])).To(Equal("�")) + _, err := io.ReadAll(reader) + Expect(err).To(MatchError(boom)) + }) }) })