mirror of
https://github.com/navidrome/navidrome.git
synced 2026-10-08 02:17:25 +02:00
Expand backend lyrics support with richer sidecar formats and upgrade the OpenSubsonic songLyrics implementation to the version 2 structured karaoke contract, while preserving version 1 behavior by default. Sidecar formats and parsing: - Add a TTML parser (core/lyrics/ttml.go): clock time, offset time, bare decimal seconds, nested timing contexts, and token-level <span> timing for word/syllable karaoke. Parses Apple Music-style metadata tracks (translation and pronunciation/transliteration) and agent metadata into per-track agents[] plus per-cue-line agentId. Hydrates missing line timing from cue timing. - Add an SRT parser (core/lyrics/srt.go). - Add a LRCLIB Lyricsfile (.yaml/.yml) parser (model/lyricsfile.go): maps per-word lines[].words[] to cues with inclusive UTF-8 byte offsets and attributes overlapping lines to synthetic voice agents so parallel vocals split correctly in the enhanced response. - Extend LRC parsing for Enhanced LRC inline <mm:ss.xx> word-timing markers. - Add UTF-8 BOM and UTF-16 LE support for TTML/LRC sidecars. - Parse the above formats from embedded tags as well as sidecar files. Source resolution: - Default lyricspriority is now ".ttml,.yaml,.yml,.elrc,.lrc,.srt,.txt,embedded" so the new formats are discoverable without manual configuration. - Preserve configured source priority across duplicate media-file candidates instead of only checking the first DB match, so higher-priority sidecar lyrics on older duplicates can still win. - Raise the embedded-lyrics tag maxLength to 1 MB to fit word-timed TTML/Enhanced-LRC karaoke for a full song. OpenSubsonic songLyrics v2: - Advertise songLyrics versions [1, 2]. - With enhanced=true, getLyricsBySongId may return structuredLyrics.kind (main/translation/pronunciation), cueLine[] line-level karaoke groupings, cueLine.cue[] timed words/syllables with required UTF-8 byteStart/byteEnd, reusable structuredLyrics.agents[], and cueLine.agentId references. - Without enhanced=true, the response stays v1-compatible: no kind, no cueLine, no agents, no non-main tracks; the existing line[] payload is always populated so legacy clients keep working. Contract details: - cueLine is emitted only for synced lyrics with cue data. - Within a cueLine, cue.end is normalized all-or-none and overlaps are removed; overlaps across separate cueLines remain valid for parallel vocal layers. - Missing cue end-times are filled from the next cue or the parent line. - When cueLines share an index, the one whose agent has role "main" is first. - LyricCue.Value is serialized as XML chardata; cues with nil start are skipped rather than serialized as 0. Refactoring: - Move pure format parsers into model/ (lyrics.go, lyrics_ttml.go, lyrics_srt.go, lyrics_embedded.go, lyricsfile.go) and extract Subsonic response building into server/subsonic/lyrics.go. - Centralize lyric-kind constants and add Lyrics.EffectiveKind/IsMainKind. - Add gg.Clone helper. Spec references: https://github.com/opensubsonic/open-subsonic-api/discussions/213 https://github.com/opensubsonic/open-subsonic-api/pull/218 (songLyrics v2) https://github.com/opensubsonic/open-subsonic-api/pull/228 (cue byte offsets)
160 lines
4.8 KiB
Go
160 lines
4.8 KiB
Go
package model
|
|
|
|
import (
|
|
"strings"
|
|
|
|
. "github.com/onsi/ginkgo/v2"
|
|
. "github.com/onsi/gomega"
|
|
)
|
|
|
|
var _ = Describe("ParseEmbedded", func() {
|
|
It("should parse embedded TTML with the tag language as the default", func() {
|
|
content := `<tt xmlns="http://www.w3.org/ns/ttml" xmlns:ttm="http://www.w3.org/ns/ttml#metadata">
|
|
<head>
|
|
<metadata>
|
|
<ttm:agent xml:id="lead" ttm:type="person">
|
|
<ttm:name>Lead Vocal</ttm:name>
|
|
</ttm:agent>
|
|
</metadata>
|
|
</head>
|
|
<body>
|
|
<div>
|
|
<p begin="00:00:01.000" end="00:00:03.000">
|
|
<span begin="00:00:01.000" end="00:00:02.000" ttm:agent="lead">Hello </span><span begin="00:00:02.000" end="00:00:03.000" ttm:agent="lead">world</span>
|
|
</p>
|
|
</div>
|
|
</body>
|
|
</tt>`
|
|
|
|
list, err := ParseEmbedded("ENG", content)
|
|
|
|
// ParseEmbedded's job is to detect TTML and apply the tag language as the
|
|
// default; the parser's cue/agent details are covered in lyrics_ttml_test.go.
|
|
Expect(err).ToNot(HaveOccurred())
|
|
Expect(list).To(HaveLen(1))
|
|
Expect(list[0].Kind).To(Equal("main"))
|
|
Expect(list[0].Lang).To(Equal("eng"))
|
|
Expect(list[0].Synced).To(BeTrue())
|
|
Expect(list[0].Line[0].Value).To(Equal("Hello world"))
|
|
})
|
|
|
|
It("should preserve embedded TTML translation and pronunciation tracks", func() {
|
|
content := `<tt xmlns="http://www.w3.org/ns/ttml" xmlns:itunes="http://music.apple.com/lyric-ttml-internal">
|
|
<head>
|
|
<metadata>
|
|
<iTunesMetadata xmlns="http://music.apple.com/lyric-ttml-internal">
|
|
<translations>
|
|
<translation xml:lang="es">
|
|
<text for="L1">Hola</text>
|
|
</translation>
|
|
</translations>
|
|
<transliterations>
|
|
<transliteration xml:lang="ja-Latn">
|
|
<text for="L1"><span begin="00:00:01.000" end="00:00:01.300" xmlns="http://www.w3.org/ns/ttml">ko</span><span begin="00:00:01.300" end="00:00:01.600" xmlns="http://www.w3.org/ns/ttml">nni</span></text>
|
|
</transliteration>
|
|
</transliterations>
|
|
</iTunesMetadata>
|
|
</metadata>
|
|
</head>
|
|
<body xml:lang="ja">
|
|
<div>
|
|
<p begin="00:00:01.000" end="00:00:02.000" itunes:key="L1">こんにちは</p>
|
|
</div>
|
|
</body>
|
|
</tt>`
|
|
|
|
list, err := ParseEmbedded("eng", content)
|
|
|
|
Expect(err).ToNot(HaveOccurred())
|
|
Expect(list).To(HaveLen(3))
|
|
Expect(list[0].Kind).To(Equal("main"))
|
|
Expect(list[0].Lang).To(Equal("ja"))
|
|
Expect(list[0].Line[0].Value).To(Equal("こんにちは"))
|
|
Expect(list[1].Kind).To(Equal("translation"))
|
|
Expect(list[1].Lang).To(Equal("es"))
|
|
Expect(list[1].Line[0].Value).To(Equal("Hola"))
|
|
Expect(list[2].Kind).To(Equal("pronunciation"))
|
|
Expect(list[2].Lang).To(Equal("ja-latn"))
|
|
Expect(list[2].Line[0].Value).To(Equal("konni"))
|
|
Expect(list[2].Line[0].Cue).To(HaveLen(2))
|
|
})
|
|
|
|
It("should parse embedded SRT with the tag language", func() {
|
|
content := `1
|
|
00:00:18,800 --> 00:00:22,800
|
|
We're from subtitles
|
|
|
|
2
|
|
00:00:22,801 --> 00:00:26,000
|
|
Another subtitle line`
|
|
|
|
list, err := ParseEmbedded("POR", content)
|
|
|
|
Expect(err).ToNot(HaveOccurred())
|
|
Expect(list).To(Equal(LyricList{
|
|
{
|
|
Lang: "por",
|
|
Line: []Line{
|
|
{
|
|
Start: new(int64(18800)),
|
|
End: new(int64(22800)),
|
|
Value: "We're from subtitles",
|
|
},
|
|
{
|
|
Start: new(int64(22801)),
|
|
End: new(int64(26000)),
|
|
Value: "Another subtitle line",
|
|
},
|
|
},
|
|
Synced: true,
|
|
},
|
|
}))
|
|
})
|
|
|
|
It("should parse embedded SRT blocks separated by whitespace-only blank lines", func() {
|
|
content := "1\n00:00:01,000 --> 00:00:02,000\nFirst subtitle\n \n2\n00:00:03,000 --> 00:00:04,000\nSecond subtitle"
|
|
|
|
list, err := ParseEmbedded("eng", content)
|
|
|
|
Expect(err).ToNot(HaveOccurred())
|
|
Expect(list).To(HaveLen(1))
|
|
Expect(list[0].Line).To(Equal([]Line{
|
|
{Start: new(int64(1000)), End: new(int64(2000)), Value: "First subtitle"},
|
|
{Start: new(int64(3000)), End: new(int64(4000)), Value: "Second subtitle"},
|
|
}))
|
|
})
|
|
|
|
It("should keep embedded enhanced LRC cues", func() {
|
|
content := "[00:01.00]<00:01.00>Lead <00:01.50>words"
|
|
|
|
list, err := ParseEmbedded("eng", content)
|
|
|
|
Expect(err).ToNot(HaveOccurred())
|
|
Expect(list).To(HaveLen(1))
|
|
Expect(list[0].Lang).To(Equal("eng"))
|
|
Expect(list[0].Synced).To(BeTrue())
|
|
Expect(list[0].Line[0].Value).To(Equal("Lead words"))
|
|
Expect(list[0].Line[0].Cue).To(HaveLen(2))
|
|
})
|
|
|
|
It("should fall back to plain lyrics when embedded TTML is invalid", func() {
|
|
content := `<tt xmlns="http://www.w3.org/ns/ttml">
|
|
<body>
|
|
<p begin="not-a-time">Broken</p>
|
|
</body>
|
|
</tt>`
|
|
|
|
list, err := ParseEmbedded("eng", content)
|
|
|
|
Expect(err).ToNot(HaveOccurred())
|
|
Expect(list).To(HaveLen(1))
|
|
Expect(list[0].Lang).To(Equal("eng"))
|
|
Expect(list[0].Synced).To(BeFalse())
|
|
Expect(list[0].Line).ToNot(BeEmpty())
|
|
values := make([]string, 0, len(list[0].Line))
|
|
for _, line := range list[0].Line {
|
|
values = append(values, line.Value)
|
|
}
|
|
Expect(strings.Join(values, "\n")).To(ContainSubstring("Broken"))
|
|
})
|
|
})
|