mirror of
https://github.com/navidrome/navidrome.git
synced 2026-10-08 02:17:25 +02:00
Expand backend lyrics support with richer sidecar formats and upgrade the OpenSubsonic songLyrics implementation to the version 2 structured karaoke contract, while preserving version 1 behavior by default. Sidecar formats and parsing: - Add a TTML parser (core/lyrics/ttml.go): clock time, offset time, bare decimal seconds, nested timing contexts, and token-level <span> timing for word/syllable karaoke. Parses Apple Music-style metadata tracks (translation and pronunciation/transliteration) and agent metadata into per-track agents[] plus per-cue-line agentId. Hydrates missing line timing from cue timing. - Add an SRT parser (core/lyrics/srt.go). - Add a LRCLIB Lyricsfile (.yaml/.yml) parser (model/lyricsfile.go): maps per-word lines[].words[] to cues with inclusive UTF-8 byte offsets and attributes overlapping lines to synthetic voice agents so parallel vocals split correctly in the enhanced response. - Extend LRC parsing for Enhanced LRC inline <mm:ss.xx> word-timing markers. - Add UTF-8 BOM and UTF-16 LE support for TTML/LRC sidecars. - Parse the above formats from embedded tags as well as sidecar files. Source resolution: - Default lyricspriority is now ".ttml,.yaml,.yml,.elrc,.lrc,.srt,.txt,embedded" so the new formats are discoverable without manual configuration. - Preserve configured source priority across duplicate media-file candidates instead of only checking the first DB match, so higher-priority sidecar lyrics on older duplicates can still win. - Raise the embedded-lyrics tag maxLength to 1 MB to fit word-timed TTML/Enhanced-LRC karaoke for a full song. OpenSubsonic songLyrics v2: - Advertise songLyrics versions [1, 2]. - With enhanced=true, getLyricsBySongId may return structuredLyrics.kind (main/translation/pronunciation), cueLine[] line-level karaoke groupings, cueLine.cue[] timed words/syllables with required UTF-8 byteStart/byteEnd, reusable structuredLyrics.agents[], and cueLine.agentId references. - Without enhanced=true, the response stays v1-compatible: no kind, no cueLine, no agents, no non-main tracks; the existing line[] payload is always populated so legacy clients keep working. Contract details: - cueLine is emitted only for synced lyrics with cue data. - Within a cueLine, cue.end is normalized all-or-none and overlaps are removed; overlaps across separate cueLines remain valid for parallel vocal layers. - Missing cue end-times are filled from the next cue or the parent line. - When cueLines share an index, the one whose agent has role "main" is first. - LyricCue.Value is serialized as XML chardata; cues with nil start are skipped rather than serialized as 0. Refactoring: - Move pure format parsers into model/ (lyrics.go, lyrics_ttml.go, lyrics_srt.go, lyrics_embedded.go, lyricsfile.go) and extract Subsonic response building into server/subsonic/lyrics.go. - Centralize lyric-kind constants and add Lyrics.EffectiveKind/IsMainKind. - Add gg.Clone helper. Spec references: https://github.com/opensubsonic/open-subsonic-api/discussions/213 https://github.com/opensubsonic/open-subsonic-api/pull/218 (songLyrics v2) https://github.com/opensubsonic/open-subsonic-api/pull/228 (cue byte offsets)
276 lines
6.9 KiB
Go
276 lines
6.9 KiB
Go
package model
|
|
|
|
import (
|
|
"fmt"
|
|
"strings"
|
|
|
|
"github.com/navidrome/navidrome/utils/str"
|
|
"gopkg.in/yaml.v3"
|
|
)
|
|
|
|
// ParseLyricsfile parses a LRCLIB Lyricsfile YAML document
|
|
// (see https://github.com/tranxuanthang/lrcget/blob/main/LYRICSFILE_CONCEPT.md)
|
|
// into a model.LyricList containing a single main Lyrics entry. Returns
|
|
// (nil, nil) when the input parses as YAML but does not declare Lyricsfile
|
|
// version 1.0.
|
|
//
|
|
// When the source contains per-word timing via lines[].words[], each word
|
|
// becomes a model.Cue with inclusive UTF-8 byte offsets into Line.Value, and
|
|
// overlapping lines are attributed to synthetic voice agents via lowest-free
|
|
// voice ID assignment so the OpenSubsonic v2 enhanced response can split
|
|
// parallel vocals.
|
|
func ParseLyricsfile(text string) (LyricList, error) {
|
|
var doc lyricsfileDocument
|
|
dec := yaml.NewDecoder(strings.NewReader(text))
|
|
dec.KnownFields(false)
|
|
if err := dec.Decode(&doc); err != nil {
|
|
return nil, fmt.Errorf("not a valid Lyricsfile YAML: %w", err)
|
|
}
|
|
|
|
if strings.TrimSpace(doc.Version) != lyricsfileVersion {
|
|
return nil, nil
|
|
}
|
|
|
|
lyrics := Lyrics{
|
|
DisplayArtist: str.SanitizeText(doc.Metadata.Artist),
|
|
DisplayTitle: str.SanitizeText(doc.Metadata.Title),
|
|
Lang: normalizeLyricLang(doc.Metadata.Language),
|
|
Kind: LyricKindMain,
|
|
}
|
|
if doc.Metadata.OffsetMs != 0 {
|
|
off := doc.Metadata.OffsetMs
|
|
lyrics.Offset = &off
|
|
}
|
|
|
|
if doc.Metadata.Instrumental {
|
|
return LyricList{NormalizeLyrics(lyrics)}, nil
|
|
}
|
|
|
|
if len(doc.Lines) == 0 {
|
|
lines := buildPlainLyricsfileLines(doc.Plain)
|
|
if len(lines) == 0 {
|
|
return nil, nil
|
|
}
|
|
lyrics.Line = lines
|
|
return LyricList{NormalizeLyrics(lyrics)}, nil
|
|
}
|
|
|
|
lines, agents := buildLyricsfileLines(doc.Lines)
|
|
lyrics.Line = lines
|
|
lyrics.Agents = agents
|
|
lyrics.Synced = true
|
|
return LyricList{NormalizeLyrics(lyrics)}, nil
|
|
}
|
|
|
|
const lyricsfileVersion = "1.0"
|
|
|
|
type lyricsfileDocument struct {
|
|
Version string `yaml:"version"`
|
|
Metadata lyricsfileMetadata `yaml:"metadata"`
|
|
Lines []lyricsfileLineEntry `yaml:"lines"`
|
|
Plain string `yaml:"plain"`
|
|
}
|
|
|
|
type lyricsfileMetadata struct {
|
|
Title string `yaml:"title"`
|
|
Artist string `yaml:"artist"`
|
|
Album string `yaml:"album"`
|
|
DurationMs int64 `yaml:"duration_ms"`
|
|
OffsetMs int64 `yaml:"offset_ms"`
|
|
Language string `yaml:"language"`
|
|
Instrumental bool `yaml:"instrumental"`
|
|
}
|
|
|
|
type lyricsfileLineEntry struct {
|
|
Text string `yaml:"text"`
|
|
StartMs int64 `yaml:"start_ms"`
|
|
EndMs *int64 `yaml:"end_ms"`
|
|
Words []lyricsfileWordEntry `yaml:"words"`
|
|
}
|
|
|
|
type lyricsfileWordEntry struct {
|
|
Text string `yaml:"text"`
|
|
StartMs int64 `yaml:"start_ms"`
|
|
EndMs *int64 `yaml:"end_ms"`
|
|
}
|
|
|
|
// buildLyricsfileLines converts YAML line entries to model.Line entries with
|
|
// per-cue AgentIDs assigned by streaming overlap clustering (lowest-free
|
|
// voice ID). The Agents slice is emitted only when at least one cue carries
|
|
// attribution AND more than one voice is used; otherwise AgentIDs are
|
|
// stripped so the wire shape stays simple per the OpenSubsonic spec rule
|
|
// "agents should not be emitted without cueLine data".
|
|
func buildLyricsfileLines(entries []lyricsfileLineEntry) ([]Line, []Agent) {
|
|
if len(entries) == 0 {
|
|
return nil, nil
|
|
}
|
|
|
|
// Resolved end timestamps per entry: explicit end_ms, final word end_ms,
|
|
// then the next entry's start. The last entry's end stays nil when no
|
|
// explicit or word-level end is available.
|
|
ends := make([]*int64, len(entries))
|
|
for i := range entries {
|
|
var nextStart *int64
|
|
if i+1 < len(entries) {
|
|
v := entries[i+1].StartMs
|
|
nextStart = &v
|
|
}
|
|
ends[i] = lyricsfileLineEnd(entries[i], nextStart)
|
|
}
|
|
|
|
active := map[int]int64{}
|
|
maxVoice := -1
|
|
anyCues := false
|
|
lines := make([]Line, 0, len(entries))
|
|
|
|
for i, entry := range entries {
|
|
for vID, vEnd := range active {
|
|
if vEnd <= entry.StartMs {
|
|
delete(active, vID)
|
|
}
|
|
}
|
|
|
|
voiceID := 0
|
|
for {
|
|
if _, busy := active[voiceID]; !busy {
|
|
break
|
|
}
|
|
voiceID++
|
|
}
|
|
if voiceID > maxVoice {
|
|
maxVoice = voiceID
|
|
}
|
|
|
|
agentID := fmt.Sprintf("voice-%d", voiceID)
|
|
cues, value := wordsToLineCues(entry, agentID)
|
|
if len(cues) > 0 {
|
|
anyCues = true
|
|
}
|
|
|
|
startMs := entry.StartMs
|
|
line := Line{
|
|
Start: &startMs,
|
|
End: ends[i],
|
|
Value: value,
|
|
Cue: cues,
|
|
}
|
|
lines = append(lines, line)
|
|
|
|
var endMs int64
|
|
if ends[i] != nil {
|
|
endMs = *ends[i]
|
|
} else {
|
|
endMs = entry.StartMs
|
|
}
|
|
active[voiceID] = endMs
|
|
}
|
|
|
|
// Monophonic source, or attribution that has nowhere to land: emit no
|
|
// agents and strip per-cue AgentIDs to keep the wire shape simple.
|
|
if maxVoice <= 0 || !anyCues {
|
|
for i := range lines {
|
|
for j := range lines[i].Cue {
|
|
lines[i].Cue[j].AgentID = ""
|
|
}
|
|
}
|
|
return lines, nil
|
|
}
|
|
|
|
agents := make([]Agent, 0, maxVoice+1)
|
|
for v := 0; v <= maxVoice; v++ {
|
|
role := "voice"
|
|
if v == 0 {
|
|
role = "main"
|
|
}
|
|
agents = append(agents, Agent{
|
|
ID: fmt.Sprintf("voice-%d", v),
|
|
Role: role,
|
|
})
|
|
}
|
|
return lines, agents
|
|
}
|
|
|
|
func lyricsfileLineEnd(entry lyricsfileLineEntry, nextStart *int64) *int64 {
|
|
if entry.EndMs != nil {
|
|
v := *entry.EndMs
|
|
return &v
|
|
}
|
|
if len(entry.Words) > 0 {
|
|
lastWord := entry.Words[len(entry.Words)-1]
|
|
if lastWord.EndMs != nil {
|
|
v := *lastWord.EndMs
|
|
return &v
|
|
}
|
|
}
|
|
if nextStart != nil {
|
|
v := *nextStart
|
|
return &v
|
|
}
|
|
return nil
|
|
}
|
|
|
|
func buildPlainLyricsfileLines(plain string) []Line {
|
|
plain = str.SanitizeText(plain)
|
|
rawLines := strings.Split(plain, "\n")
|
|
lines := make([]Line, 0, len(rawLines))
|
|
for _, raw := range rawLines {
|
|
value := strings.TrimSpace(raw)
|
|
if value == "" {
|
|
continue
|
|
}
|
|
lines = append(lines, Line{Value: value})
|
|
}
|
|
return lines
|
|
}
|
|
|
|
// wordsToLineCues converts a Lyricsfile line entry's words[] into model.Cue
|
|
// entries with inclusive UTF-8 byte offsets into the reconstructed line
|
|
// value. The line value is built from cue text concatenation rather than
|
|
// trusting entry.Text, because the Lyricsfile spec only requires word.text
|
|
// to "approximate" line.text - byte offsets must always land inside
|
|
// Line.Value.
|
|
func wordsToLineCues(entry lyricsfileLineEntry, agentID string) ([]Cue, string) {
|
|
if len(entry.Words) == 0 {
|
|
return nil, str.SanitizeText(entry.Text)
|
|
}
|
|
|
|
var sb strings.Builder
|
|
for _, w := range entry.Words {
|
|
sb.WriteString(w.Text)
|
|
}
|
|
lineValue := sb.String()
|
|
|
|
cues := make([]Cue, len(entry.Words))
|
|
cursor := 0
|
|
for i, w := range entry.Words {
|
|
valueBytes := len(w.Text)
|
|
bs := cursor
|
|
be := bs
|
|
if valueBytes > 0 {
|
|
be = bs + valueBytes - 1
|
|
cursor = be + 1
|
|
}
|
|
|
|
s := w.StartMs
|
|
cue := Cue{
|
|
Start: &s,
|
|
Value: w.Text,
|
|
ByteStart: bs,
|
|
ByteEnd: be,
|
|
AgentID: agentID,
|
|
}
|
|
if w.EndMs != nil {
|
|
e := *w.EndMs
|
|
cue.End = &e
|
|
}
|
|
cues[i] = cue
|
|
}
|
|
|
|
for i := 0; i < len(cues)-1; i++ {
|
|
if cues[i].End == nil && cues[i+1].Start != nil {
|
|
v := *cues[i+1].Start
|
|
cues[i].End = &v
|
|
}
|
|
}
|
|
return cues, lineValue
|
|
}
|