navidrome/model/lyricsfile.go
Yuuta 3a14faa033
feat(subsonic): add structured sidecar lyrics support with OpenSubsonic v2 karaoke cues and agent layers (#5076)
Expand backend lyrics support with richer sidecar formats and upgrade the
OpenSubsonic songLyrics implementation to the version 2 structured karaoke
contract, while preserving version 1 behavior by default.

Sidecar formats and parsing:
- Add a TTML parser (core/lyrics/ttml.go): clock time, offset time, bare
  decimal seconds, nested timing contexts, and token-level <span> timing for
  word/syllable karaoke. Parses Apple Music-style metadata tracks (translation
  and pronunciation/transliteration) and agent metadata into per-track agents[]
  plus per-cue-line agentId. Hydrates missing line timing from cue timing.
- Add an SRT parser (core/lyrics/srt.go).
- Add a LRCLIB Lyricsfile (.yaml/.yml) parser (model/lyricsfile.go): maps
  per-word lines[].words[] to cues with inclusive UTF-8 byte offsets and
  attributes overlapping lines to synthetic voice agents so parallel vocals
  split correctly in the enhanced response.
- Extend LRC parsing for Enhanced LRC inline <mm:ss.xx> word-timing markers.
- Add UTF-8 BOM and UTF-16 LE support for TTML/LRC sidecars.
- Parse the above formats from embedded tags as well as sidecar files.

Source resolution:
- Default lyricspriority is now
  ".ttml,.yaml,.yml,.elrc,.lrc,.srt,.txt,embedded" so the new formats are
  discoverable without manual configuration.
- Preserve configured source priority across duplicate media-file candidates
  instead of only checking the first DB match, so higher-priority sidecar
  lyrics on older duplicates can still win.
- Raise the embedded-lyrics tag maxLength to 1 MB to fit word-timed
  TTML/Enhanced-LRC karaoke for a full song.

OpenSubsonic songLyrics v2:
- Advertise songLyrics versions [1, 2].
- With enhanced=true, getLyricsBySongId may return structuredLyrics.kind
  (main/translation/pronunciation), cueLine[] line-level karaoke groupings,
  cueLine.cue[] timed words/syllables with required UTF-8 byteStart/byteEnd,
  reusable structuredLyrics.agents[], and cueLine.agentId references.
- Without enhanced=true, the response stays v1-compatible: no kind, no cueLine,
  no agents, no non-main tracks; the existing line[] payload is always
  populated so legacy clients keep working.

Contract details:
- cueLine is emitted only for synced lyrics with cue data.
- Within a cueLine, cue.end is normalized all-or-none and overlaps are removed;
  overlaps across separate cueLines remain valid for parallel vocal layers.
- Missing cue end-times are filled from the next cue or the parent line.
- When cueLines share an index, the one whose agent has role "main" is first.
- LyricCue.Value is serialized as XML chardata; cues with nil start are skipped
  rather than serialized as 0.

Refactoring:
- Move pure format parsers into model/ (lyrics.go, lyrics_ttml.go,
  lyrics_srt.go, lyrics_embedded.go, lyricsfile.go) and extract Subsonic
  response building into server/subsonic/lyrics.go.
- Centralize lyric-kind constants and add Lyrics.EffectiveKind/IsMainKind.
- Add gg.Clone helper.

Spec references:
  https://github.com/opensubsonic/open-subsonic-api/discussions/213
  https://github.com/opensubsonic/open-subsonic-api/pull/218 (songLyrics v2)
  https://github.com/opensubsonic/open-subsonic-api/pull/228 (cue byte offsets)
2026-06-19 12:00:58 -04:00

276 lines
6.9 KiB
Go

package model
import (
"fmt"
"strings"
"github.com/navidrome/navidrome/utils/str"
"gopkg.in/yaml.v3"
)
// ParseLyricsfile parses a LRCLIB Lyricsfile YAML document
// (see https://github.com/tranxuanthang/lrcget/blob/main/LYRICSFILE_CONCEPT.md)
// into a model.LyricList containing a single main Lyrics entry. Returns
// (nil, nil) when the input parses as YAML but does not declare Lyricsfile
// version 1.0.
//
// When the source contains per-word timing via lines[].words[], each word
// becomes a model.Cue with inclusive UTF-8 byte offsets into Line.Value, and
// overlapping lines are attributed to synthetic voice agents via lowest-free
// voice ID assignment so the OpenSubsonic v2 enhanced response can split
// parallel vocals.
func ParseLyricsfile(text string) (LyricList, error) {
var doc lyricsfileDocument
dec := yaml.NewDecoder(strings.NewReader(text))
dec.KnownFields(false)
if err := dec.Decode(&doc); err != nil {
return nil, fmt.Errorf("not a valid Lyricsfile YAML: %w", err)
}
if strings.TrimSpace(doc.Version) != lyricsfileVersion {
return nil, nil
}
lyrics := Lyrics{
DisplayArtist: str.SanitizeText(doc.Metadata.Artist),
DisplayTitle: str.SanitizeText(doc.Metadata.Title),
Lang: normalizeLyricLang(doc.Metadata.Language),
Kind: LyricKindMain,
}
if doc.Metadata.OffsetMs != 0 {
off := doc.Metadata.OffsetMs
lyrics.Offset = &off
}
if doc.Metadata.Instrumental {
return LyricList{NormalizeLyrics(lyrics)}, nil
}
if len(doc.Lines) == 0 {
lines := buildPlainLyricsfileLines(doc.Plain)
if len(lines) == 0 {
return nil, nil
}
lyrics.Line = lines
return LyricList{NormalizeLyrics(lyrics)}, nil
}
lines, agents := buildLyricsfileLines(doc.Lines)
lyrics.Line = lines
lyrics.Agents = agents
lyrics.Synced = true
return LyricList{NormalizeLyrics(lyrics)}, nil
}
const lyricsfileVersion = "1.0"
type lyricsfileDocument struct {
Version string `yaml:"version"`
Metadata lyricsfileMetadata `yaml:"metadata"`
Lines []lyricsfileLineEntry `yaml:"lines"`
Plain string `yaml:"plain"`
}
type lyricsfileMetadata struct {
Title string `yaml:"title"`
Artist string `yaml:"artist"`
Album string `yaml:"album"`
DurationMs int64 `yaml:"duration_ms"`
OffsetMs int64 `yaml:"offset_ms"`
Language string `yaml:"language"`
Instrumental bool `yaml:"instrumental"`
}
type lyricsfileLineEntry struct {
Text string `yaml:"text"`
StartMs int64 `yaml:"start_ms"`
EndMs *int64 `yaml:"end_ms"`
Words []lyricsfileWordEntry `yaml:"words"`
}
type lyricsfileWordEntry struct {
Text string `yaml:"text"`
StartMs int64 `yaml:"start_ms"`
EndMs *int64 `yaml:"end_ms"`
}
// buildLyricsfileLines converts YAML line entries to model.Line entries with
// per-cue AgentIDs assigned by streaming overlap clustering (lowest-free
// voice ID). The Agents slice is emitted only when at least one cue carries
// attribution AND more than one voice is used; otherwise AgentIDs are
// stripped so the wire shape stays simple per the OpenSubsonic spec rule
// "agents should not be emitted without cueLine data".
func buildLyricsfileLines(entries []lyricsfileLineEntry) ([]Line, []Agent) {
if len(entries) == 0 {
return nil, nil
}
// Resolved end timestamps per entry: explicit end_ms, final word end_ms,
// then the next entry's start. The last entry's end stays nil when no
// explicit or word-level end is available.
ends := make([]*int64, len(entries))
for i := range entries {
var nextStart *int64
if i+1 < len(entries) {
v := entries[i+1].StartMs
nextStart = &v
}
ends[i] = lyricsfileLineEnd(entries[i], nextStart)
}
active := map[int]int64{}
maxVoice := -1
anyCues := false
lines := make([]Line, 0, len(entries))
for i, entry := range entries {
for vID, vEnd := range active {
if vEnd <= entry.StartMs {
delete(active, vID)
}
}
voiceID := 0
for {
if _, busy := active[voiceID]; !busy {
break
}
voiceID++
}
if voiceID > maxVoice {
maxVoice = voiceID
}
agentID := fmt.Sprintf("voice-%d", voiceID)
cues, value := wordsToLineCues(entry, agentID)
if len(cues) > 0 {
anyCues = true
}
startMs := entry.StartMs
line := Line{
Start: &startMs,
End: ends[i],
Value: value,
Cue: cues,
}
lines = append(lines, line)
var endMs int64
if ends[i] != nil {
endMs = *ends[i]
} else {
endMs = entry.StartMs
}
active[voiceID] = endMs
}
// Monophonic source, or attribution that has nowhere to land: emit no
// agents and strip per-cue AgentIDs to keep the wire shape simple.
if maxVoice <= 0 || !anyCues {
for i := range lines {
for j := range lines[i].Cue {
lines[i].Cue[j].AgentID = ""
}
}
return lines, nil
}
agents := make([]Agent, 0, maxVoice+1)
for v := 0; v <= maxVoice; v++ {
role := "voice"
if v == 0 {
role = "main"
}
agents = append(agents, Agent{
ID: fmt.Sprintf("voice-%d", v),
Role: role,
})
}
return lines, agents
}
func lyricsfileLineEnd(entry lyricsfileLineEntry, nextStart *int64) *int64 {
if entry.EndMs != nil {
v := *entry.EndMs
return &v
}
if len(entry.Words) > 0 {
lastWord := entry.Words[len(entry.Words)-1]
if lastWord.EndMs != nil {
v := *lastWord.EndMs
return &v
}
}
if nextStart != nil {
v := *nextStart
return &v
}
return nil
}
func buildPlainLyricsfileLines(plain string) []Line {
plain = str.SanitizeText(plain)
rawLines := strings.Split(plain, "\n")
lines := make([]Line, 0, len(rawLines))
for _, raw := range rawLines {
value := strings.TrimSpace(raw)
if value == "" {
continue
}
lines = append(lines, Line{Value: value})
}
return lines
}
// wordsToLineCues converts a Lyricsfile line entry's words[] into model.Cue
// entries with inclusive UTF-8 byte offsets into the reconstructed line
// value. The line value is built from cue text concatenation rather than
// trusting entry.Text, because the Lyricsfile spec only requires word.text
// to "approximate" line.text - byte offsets must always land inside
// Line.Value.
func wordsToLineCues(entry lyricsfileLineEntry, agentID string) ([]Cue, string) {
if len(entry.Words) == 0 {
return nil, str.SanitizeText(entry.Text)
}
var sb strings.Builder
for _, w := range entry.Words {
sb.WriteString(w.Text)
}
lineValue := sb.String()
cues := make([]Cue, len(entry.Words))
cursor := 0
for i, w := range entry.Words {
valueBytes := len(w.Text)
bs := cursor
be := bs
if valueBytes > 0 {
be = bs + valueBytes - 1
cursor = be + 1
}
s := w.StartMs
cue := Cue{
Start: &s,
Value: w.Text,
ByteStart: bs,
ByteEnd: be,
AgentID: agentID,
}
if w.EndMs != nil {
e := *w.EndMs
cue.End = &e
}
cues[i] = cue
}
for i := 0; i < len(cues)-1; i++ {
if cues[i].End == nil && cues[i+1].Start != nil {
v := *cues[i+1].Start
cues[i].End = &v
}
}
return cues, lineValue
}