diff --git a/README.md b/README.md
index 0ae5bdfaf..efecfdbcb 100644
--- a/README.md
+++ b/README.md
@@ -52,6 +52,7 @@ A share of the revenue helps fund the development of Navidrome at no additional
- **Multi-platform**, runs on macOS, Linux and Windows. **Docker** images are also provided
- Ready to use binaries for all major platforms, including **Raspberry Pi**
- Automatically **monitors your library** for changes, importing new files and reloading new metadata
+ - Supports lyrics from sidecar **.ttml**, **.elrc**, **.lrc**, **.srt**, **.txt** files and embedded **TTML**, **Enhanced LRC**, **LRC**, **SRT**, and plain-text tags (via `lyricspriority`)
- **Themeable**, modern and responsive **Web interface** based on [Material UI](https://material-ui.com)
- **Compatible** with all Subsonic/Madsonic/Airsonic [clients](https://www.navidrome.org/docs/overview/#apps)
- **Transcoding** on the fly. Can be set per user/player. **Opus encoding is supported**
diff --git a/conf/configuration.go b/conf/configuration.go
index 08f12fc94..219b7b0bd 100644
--- a/conf/configuration.go
+++ b/conf/configuration.go
@@ -776,7 +776,7 @@ func setViperDefaults() {
viper.SetDefault("artistartpriority", "artist.*, album/artist.*, external")
viper.SetDefault("artistimagefolder", "")
viper.SetDefault("discartpriority", "disc*.*, cd*.*, cover.*, folder.*, front.*, discsubtitle, embedded")
- viper.SetDefault("lyricspriority", ".lrc,.txt,embedded")
+ viper.SetDefault("lyricspriority", ".ttml,.elrc,.lrc,.srt,.txt,embedded")
viper.SetDefault("enablegravatar", false)
viper.SetDefault("enablefavourites", true)
viper.SetDefault("enablestarrating", true)
diff --git a/core/lyrics/embedded.go b/core/lyrics/embedded.go
new file mode 100644
index 000000000..4cdd898ee
--- /dev/null
+++ b/core/lyrics/embedded.go
@@ -0,0 +1,64 @@
+package lyrics
+
+import (
+ "encoding/xml"
+ "strings"
+
+ "github.com/navidrome/navidrome/log"
+ "github.com/navidrome/navidrome/model"
+)
+
+// ParseEmbedded parses lyrics read from media-file metadata tags. It detects rich
+// payloads before falling back to the generic LRC/plain-text parser, because
+// text sanitization would otherwise strip TTML XML markup.
+func ParseEmbedded(language, text string) (model.LyricList, error) {
+ text = strings.TrimPrefix(text, "\ufeff")
+
+ if isTTMLDocument(text) {
+ list, err := parseTTMLWithDefaultLang([]byte(text), language)
+ if err == nil && len(list) > 0 {
+ return list, nil
+ }
+ if err != nil {
+ log.Warn("Error parsing embedded TTML lyrics, falling back to plain lyrics", "error", err)
+ }
+ }
+
+ list, err := parseSRTWithLanguage([]byte(text), language)
+ if err == nil && len(list) > 0 {
+ return list, nil
+ }
+ if err != nil && strings.Contains(text, "-->") {
+ log.Warn("Error parsing embedded SRT lyrics, falling back to plain lyrics", "error", err)
+ }
+
+ lyric, err := model.ToLyrics(language, text)
+ if err != nil {
+ return nil, err
+ }
+ if lyric == nil || lyric.IsEmpty() {
+ return nil, nil
+ }
+ return model.LyricList{*lyric}, nil
+}
+
+func isTTMLDocument(text string) bool {
+ decoder := xml.NewDecoder(strings.NewReader(strings.TrimSpace(text)))
+ for {
+ token, err := decoder.Token()
+ if err != nil {
+ return false
+ }
+ if start, ok := token.(xml.StartElement); ok {
+ return strings.EqualFold(start.Name.Local, "tt")
+ }
+ }
+}
+
+func normalizeEmbeddedLanguage(language string) string {
+ language = strings.ToLower(strings.TrimSpace(language))
+ if language == "" {
+ return "xxx"
+ }
+ return language
+}
diff --git a/core/lyrics/embedded_test.go b/core/lyrics/embedded_test.go
new file mode 100644
index 000000000..d875bbdcf
--- /dev/null
+++ b/core/lyrics/embedded_test.go
@@ -0,0 +1,169 @@
+package lyrics
+
+import (
+ "strings"
+
+ "github.com/navidrome/navidrome/model"
+ . "github.com/onsi/ginkgo/v2"
+ . "github.com/onsi/gomega"
+)
+
+var _ = Describe("ParseEmbedded", func() {
+ It("should parse embedded TTML with the tag language as the default", func() {
+ content := `
+
+
+
+ Lead Vocal
+
+
+
+
+
+
+`
+
+ list, err := ParseEmbedded("ENG", content)
+
+ Expect(err).ToNot(HaveOccurred())
+ Expect(list).To(HaveLen(1))
+ Expect(list[0].Kind).To(Equal("main"))
+ Expect(list[0].Lang).To(Equal("eng"))
+ Expect(list[0].Synced).To(BeTrue())
+ Expect(list[0].Agents).To(Equal([]model.Agent{{ID: "lead", Role: "main", Name: "Lead Vocal"}}))
+ Expect(list[0].Line).To(HaveLen(1))
+ Expect(list[0].Line[0].Start).To(Equal(ptr(int64(1000))))
+ Expect(list[0].Line[0].End).To(Equal(ptr(int64(3000))))
+ Expect(list[0].Line[0].Value).To(Equal("Hello world"))
+ Expect(list[0].Line[0].Cue).To(HaveLen(2))
+ Expect(list[0].Line[0].Cue[0].AgentID).To(Equal("lead"))
+ Expect(list[0].Line[0].Cue[0].ByteStart).To(Equal(0))
+ Expect(list[0].Line[0].Cue[0].ByteEnd).To(Equal(5))
+ Expect(list[0].Line[0].Cue[1].ByteStart).To(Equal(6))
+ Expect(list[0].Line[0].Cue[1].ByteEnd).To(Equal(10))
+ })
+
+ It("should preserve embedded TTML translation and pronunciation tracks", func() {
+ content := `
+
+
+
+
+
+ Hola
+
+
+
+
+ konni
+
+
+
+
+
+
+
+
+`
+
+ list, err := ParseEmbedded("eng", content)
+
+ Expect(err).ToNot(HaveOccurred())
+ Expect(list).To(HaveLen(3))
+ Expect(list[0].Kind).To(Equal("main"))
+ Expect(list[0].Lang).To(Equal("ja"))
+ Expect(list[0].Line[0].Value).To(Equal("こんにちは"))
+ Expect(list[1].Kind).To(Equal("translation"))
+ Expect(list[1].Lang).To(Equal("es"))
+ Expect(list[1].Line[0].Value).To(Equal("Hola"))
+ Expect(list[2].Kind).To(Equal("pronunciation"))
+ Expect(list[2].Lang).To(Equal("ja-latn"))
+ Expect(list[2].Line[0].Value).To(Equal("konni"))
+ Expect(list[2].Line[0].Cue).To(HaveLen(2))
+ })
+
+ It("should parse embedded SRT with the tag language", func() {
+ content := `1
+00:00:18,800 --> 00:00:22,800
+We're from subtitles
+
+2
+00:00:22,801 --> 00:00:26,000
+Another subtitle line`
+
+ list, err := ParseEmbedded("POR", content)
+
+ Expect(err).ToNot(HaveOccurred())
+ Expect(list).To(Equal(model.LyricList{
+ {
+ Lang: "por",
+ Line: []model.Line{
+ {
+ Start: ptr(int64(18800)),
+ End: ptr(int64(22800)),
+ Value: "We're from subtitles",
+ },
+ {
+ Start: ptr(int64(22801)),
+ End: ptr(int64(26000)),
+ Value: "Another subtitle line",
+ },
+ },
+ Synced: true,
+ },
+ }))
+ })
+
+ It("should parse embedded SRT blocks separated by whitespace-only blank lines", func() {
+ content := "1\n00:00:01,000 --> 00:00:02,000\nFirst subtitle\n \n2\n00:00:03,000 --> 00:00:04,000\nSecond subtitle"
+
+ list, err := ParseEmbedded("eng", content)
+
+ Expect(err).ToNot(HaveOccurred())
+ Expect(list).To(HaveLen(1))
+ Expect(list[0].Line).To(Equal([]model.Line{
+ {Start: ptr(int64(1000)), End: ptr(int64(2000)), Value: "First subtitle"},
+ {Start: ptr(int64(3000)), End: ptr(int64(4000)), Value: "Second subtitle"},
+ }))
+ })
+
+ It("should keep embedded enhanced LRC cues", func() {
+ content := "[00:01.00]<00:01.00>Lead <00:01.50>words"
+
+ list, err := ParseEmbedded("eng", content)
+
+ Expect(err).ToNot(HaveOccurred())
+ Expect(list).To(HaveLen(1))
+ Expect(list[0].Lang).To(Equal("eng"))
+ Expect(list[0].Synced).To(BeTrue())
+ Expect(list[0].Line[0].Value).To(Equal("Lead words"))
+ Expect(list[0].Line[0].Cue).To(HaveLen(2))
+ })
+
+ It("should fall back to plain lyrics when embedded TTML is invalid", func() {
+ content := `
+
+
Broken
+
+`
+
+ list, err := ParseEmbedded("eng", content)
+
+ Expect(err).ToNot(HaveOccurred())
+ Expect(list).To(HaveLen(1))
+ Expect(list[0].Lang).To(Equal("eng"))
+ Expect(list[0].Synced).To(BeFalse())
+ Expect(list[0].Line).ToNot(BeEmpty())
+ values := make([]string, 0, len(list[0].Line))
+ for _, line := range list[0].Line {
+ values = append(values, line.Value)
+ }
+ Expect(strings.Join(values, "\n")).To(ContainSubstring("Broken"))
+ })
+})
diff --git a/core/lyrics/external_test_helpers_test.go b/core/lyrics/external_test_helpers_test.go
new file mode 100644
index 000000000..6fd41aba4
--- /dev/null
+++ b/core/lyrics/external_test_helpers_test.go
@@ -0,0 +1,5 @@
+package lyrics_test
+
+func ptr[T any](v T) *T {
+ return &v
+}
diff --git a/core/lyrics/lyrics.go b/core/lyrics/lyrics.go
index 758053042..73a0479f3 100644
--- a/core/lyrics/lyrics.go
+++ b/core/lyrics/lyrics.go
@@ -14,6 +14,12 @@ type Lyrics interface {
GetLyrics(ctx context.Context, mf *model.MediaFile) (model.LyricList, error)
}
+// BatchLyrics can resolve lyrics across multiple candidate media files while
+// still honoring the configured source priority globally.
+type BatchLyrics interface {
+ GetLyricsForMediaFiles(ctx context.Context, mediaFiles []model.MediaFile) (model.LyricList, error)
+}
+
// PluginLoader discovers and loads lyrics provider plugins.
type PluginLoader interface {
LoadLyricsProvider(name string) (Lyrics, bool)
@@ -32,28 +38,53 @@ func NewLyrics(pluginLoader PluginLoader) Lyrics {
// GetLyrics returns lyrics for the given media file, trying sources in the
// order specified by conf.Server.LyricsPriority.
func (l *lyricsService) GetLyrics(ctx context.Context, mf *model.MediaFile) (model.LyricList, error) {
- var lyricsList model.LyricList
- var err error
+ return l.getLyricsForCandidates(ctx, []*model.MediaFile{mf})
+}
+// GetLyricsForMediaFiles resolves lyrics across duplicate media files while
+// preserving the configured source priority across the full candidate set.
+func (l *lyricsService) GetLyricsForMediaFiles(ctx context.Context, mediaFiles []model.MediaFile) (model.LyricList, error) {
+ candidates := make([]*model.MediaFile, 0, len(mediaFiles))
+ for i := range mediaFiles {
+ candidates = append(candidates, &mediaFiles[i])
+ }
+ return l.getLyricsForCandidates(ctx, candidates)
+}
+
+func (l *lyricsService) getLyricsForCandidates(ctx context.Context, mediaFiles []*model.MediaFile) (model.LyricList, error) {
for pattern := range strings.SplitSeq(conf.Server.LyricsPriority, ",") {
pattern = strings.TrimSpace(pattern)
- switch {
- case strings.EqualFold(pattern, "embedded"):
- lyricsList, err = fromEmbedded(ctx, mf)
- case strings.HasPrefix(pattern, "."):
- lyricsList, err = fromExternalFile(ctx, mf, strings.ToLower(pattern))
- default:
- lyricsList, err = l.fromPlugin(ctx, mf, pattern)
+ if pattern == "" {
+ continue
}
- if err != nil {
- log.Error(ctx, "error getting lyrics", "source", pattern, err)
- }
+ for _, mf := range mediaFiles {
+ if mf == nil {
+ continue
+ }
- if len(lyricsList) > 0 {
- return lyricsList, nil
+ lyricsList, err := l.getLyricsFromSource(ctx, mf, pattern)
+ if err != nil {
+ log.Error(ctx, "error getting lyrics", "source", pattern, err)
+ continue
+ }
+
+ if len(lyricsList) > 0 {
+ return lyricsList, nil
+ }
}
}
return nil, nil
}
+
+func (l *lyricsService) getLyricsFromSource(ctx context.Context, mf *model.MediaFile, pattern string) (model.LyricList, error) {
+ switch {
+ case strings.EqualFold(pattern, "embedded"):
+ return fromEmbedded(ctx, mf)
+ case strings.HasPrefix(pattern, "."):
+ return fromExternalFile(ctx, mf, pattern)
+ default:
+ return l.fromPlugin(ctx, mf, pattern)
+ }
+}
diff --git a/core/lyrics/lyrics_test.go b/core/lyrics/lyrics_test.go
index 9ab732ad1..68a7f844d 100644
--- a/core/lyrics/lyrics_test.go
+++ b/core/lyrics/lyrics_test.go
@@ -5,6 +5,7 @@ import (
"encoding/json"
"fmt"
"os"
+ "path/filepath"
"github.com/navidrome/navidrome/conf"
"github.com/navidrome/navidrome/conf/configtest"
@@ -44,6 +45,71 @@ var _ = Describe("sources", func() {
},
}
+ elrcLyrics := model.LyricList{
+ model.Lyrics{
+ DisplayArtist: "ELRC Artist",
+ DisplayTitle: "ELRC Song",
+ Lang: "eng",
+ Line: []model.Line{
+ {
+ Start: ptr(int64(1000)),
+ End: ptr(int64(3000)),
+ Value: "Lead words",
+ Cue: []model.Cue{
+ {
+ Start: ptr(int64(1000)),
+ End: ptr(int64(1500)),
+ Value: "Lead ",
+ ByteStart: 0,
+ ByteEnd: 4,
+ },
+ {
+ Start: ptr(int64(1500)),
+ End: ptr(int64(3000)),
+ Value: "words",
+ ByteStart: 5,
+ ByteEnd: 9,
+ },
+ },
+ },
+ {
+ Start: ptr(int64(3000)),
+ Value: "Fallback line",
+ },
+ },
+ Synced: true,
+ },
+ }
+
+ ttmlLyrics := model.LyricList{
+ model.Lyrics{
+ Kind: "main",
+ Lang: "eng",
+ Line: []model.Line{
+ {
+ Start: ptr(int64(18800)),
+ Value: "We're no strangers to love",
+ },
+ {
+ Start: ptr(int64(22800)),
+ Value: "You know the rules and so do I",
+ },
+ },
+ Synced: true,
+ },
+ model.Lyrics{
+ Kind: "main",
+ Lang: "por",
+ Line: []model.Line{
+ {
+ Start: ptr(int64(18800)),
+ Value: "Nao somos estranhos ao amor",
+ },
+ },
+ Synced: true,
+ },
+ }
+
unsyncedLyrics := model.LyricList{
model.Lyrics{
Lang: "xxx",
@@ -59,6 +125,25 @@ var _ = Describe("sources", func() {
},
}
+ srtLyrics := model.LyricList{
+ model.Lyrics{
+ Lang: "xxx",
+ Line: []model.Line{
+ {
+ Start: ptr(int64(18800)),
+ End: ptr(int64(22800)),
+ Value: "We're from subtitles",
+ },
+ {
+ Start: ptr(int64(22801)),
+ End: ptr(int64(26000)),
+ Value: "Another subtitle line",
+ },
+ },
+ Synced: true,
+ },
+ }
+
BeforeEach(func() {
DeferCleanup(configtest.SetupConfig())
@@ -80,7 +165,64 @@ var _ = Describe("sources", func() {
},
Entry("embedded > lrc > txt", "embedded,.lrc,.txt", embeddedLyrics),
Entry("lrc > embedded > txt", ".lrc,embedded,.txt", syncedLyrics),
- Entry("txt > lrc > embedded", ".txt,.lrc,embedded", unsyncedLyrics))
+ Entry("elrc > lrc > embedded", ".elrc,.lrc,embedded", elrcLyrics),
+ Entry("srt > txt > embedded", ".srt,.txt,embedded", srtLyrics),
+ Entry("txt > lrc > embedded", ".txt,.lrc,embedded", unsyncedLyrics),
+ Entry("ttml > elrc > lrc > srt > embedded", ".ttml,.elrc,.lrc,.srt,embedded", ttmlLyrics))
+
+ It("resolves source priority across duplicate media files", func() {
+ conf.Server.LyricsPriority = ".ttml,embedded"
+ embeddedJSON, err := json.Marshal(embeddedLyrics)
+ Expect(err).To(BeNil())
+
+ svc := lyrics.NewLyrics(nil)
+ batchSvc, ok := svc.(lyrics.BatchLyrics)
+ Expect(ok).To(BeTrue())
+
+ list, err := batchSvc.GetLyricsForMediaFiles(ctx, []model.MediaFile{
+ {
+ Lyrics: string(embeddedJSON),
+ Path: "tests/fixtures/01 Invisible (RED) Edit Version.mp3",
+ },
+ {
+ Lyrics: "[]",
+ Path: "tests/fixtures/test.mp3",
+ },
+ })
+ Expect(err).To(BeNil())
+ Expect(list).To(Equal(ttmlLyrics))
+ })
+
+ It("preserves configured sidecar suffix casing on case-sensitive filesystems", func() {
+ dir, err := os.MkdirTemp("", "lyrics-case-*")
+ Expect(err).ToNot(HaveOccurred())
+ DeferCleanup(func() {
+ Expect(os.RemoveAll(dir)).To(Succeed())
+ })
+
+ probe := filepath.Join(dir, "CASECHECK")
+ Expect(os.WriteFile(probe, []byte("probe"), 0644)).To(Succeed())
+ _, err = os.Stat(filepath.Join(dir, "casecheck"))
+ if err == nil {
+ Skip("filesystem is case-insensitive")
+ }
+ Expect(os.IsNotExist(err)).To(BeTrue())
+
+ conf.Server.LyricsPriority = ".LRC"
+ Expect(os.WriteFile(filepath.Join(dir, "song.LRC"), []byte("[00:01.00]Upper suffix"), 0644)).To(Succeed())
+
+ svc := lyrics.NewLyrics(nil)
+ list, err := svc.GetLyrics(ctx, &model.MediaFile{
+ LibraryPath: dir,
+ Path: "song.mp3",
+ })
+
+ Expect(err).To(BeNil())
+ Expect(list).To(HaveLen(1))
+ Expect(list[0].Line).To(Equal([]model.Line{
+ {Start: ptr(int64(1000)), Value: "Upper suffix"},
+ }))
+ })
Context("Errors", func() {
var RegularUserContext = XContext
diff --git a/core/lyrics/sources.go b/core/lyrics/sources.go
index 82a10ca41..7586c944f 100644
--- a/core/lyrics/sources.go
+++ b/core/lyrics/sources.go
@@ -5,6 +5,7 @@ import (
"errors"
"os"
"path"
+ "strings"
"github.com/navidrome/navidrome/log"
"github.com/navidrome/navidrome/model"
@@ -36,18 +37,38 @@ func fromExternalFile(ctx context.Context, mf *model.MediaFile, suffix string) (
return nil, err
}
- lyrics, err := model.ToLyrics("xxx", string(contents))
- if err != nil {
- log.Error(ctx, "error parsing lyric external file", "path", externalLyric, err)
- return nil, err
- } else if lyrics == nil {
+ var list model.LyricList
+ switch {
+ case strings.EqualFold(suffix, ".ttml"):
+ list, err = parseTTML(contents)
+ if err != nil {
+ log.Error(ctx, "error parsing ttml external file", "path", externalLyric, err)
+ return nil, err
+ }
+ case strings.EqualFold(suffix, ".srt"):
+ list, err = parseSRT(contents)
+ if err != nil {
+ log.Error(ctx, "error parsing srt external file", "path", externalLyric, err)
+ return nil, err
+ }
+ default:
+ lyrics, err := model.ToLyrics("xxx", string(contents))
+ if err != nil {
+ log.Error(ctx, "error parsing lyric external file", "path", externalLyric, err)
+ return nil, err
+ }
+ if lyrics != nil {
+ list = model.LyricList{*lyrics}
+ }
+ }
+
+ if len(list) == 0 {
log.Trace(ctx, "empty lyrics from external file", "path", externalLyric)
return nil, nil
}
log.Trace(ctx, "retrieved lyrics from external file", "path", externalLyric)
-
- return model.LyricList{*lyrics}, nil
+ return list, nil
}
// fromPlugin attempts to load lyrics from a plugin with the given name.
diff --git a/core/lyrics/sources_test.go b/core/lyrics/sources_test.go
index d1aefcb5d..7ca4a0a45 100644
--- a/core/lyrics/sources_test.go
+++ b/core/lyrics/sources_test.go
@@ -87,6 +87,89 @@ var _ = Describe("sources", func() {
}))
})
+ It("should return Enhanced LRC lyrics with word-level cues from a file", func() {
+ mf := model.MediaFile{Path: "tests/fixtures/test-enhanced.mp3"}
+ lyrics, err := fromExternalFile(ctx, &mf, ".lrc")
+
+ Expect(err).To(BeNil())
+ Expect(lyrics).To(HaveLen(1))
+ Expect(lyrics[0].DisplayArtist).To(Equal("Test Artist"))
+ Expect(lyrics[0].DisplayTitle).To(Equal("Enhanced Test"))
+ Expect(lyrics[0].Lang).To(Equal("eng"))
+ Expect(lyrics[0].Synced).To(BeTrue())
+ Expect(lyrics[0].Line).To(HaveLen(3))
+
+ // Line 1: has inline markers → Cue array populated
+ Expect(lyrics[0].Line[0].Start).To(Equal(ptr(int64(1000))))
+ Expect(lyrics[0].Line[0].End).To(Equal(ptr(int64(3000))))
+ Expect(lyrics[0].Line[0].Value).To(Equal("Some lyrics here"))
+ Expect(lyrics[0].Line[0].Cue).To(HaveLen(3))
+ Expect(*lyrics[0].Line[0].Cue[0].Start).To(Equal(int64(1000)))
+ Expect(lyrics[0].Line[0].Cue[0].Value).To(Equal("Some "))
+ Expect(lyrics[0].Line[0].Cue[0].End).To(Equal(ptr(int64(1500))))
+ Expect(lyrics[0].Line[0].Cue[0].ByteStart).To(Equal(0))
+ Expect(lyrics[0].Line[0].Cue[0].ByteEnd).To(Equal(4))
+ Expect(*lyrics[0].Line[0].Cue[1].Start).To(Equal(int64(1500)))
+ Expect(lyrics[0].Line[0].Cue[1].Value).To(Equal("lyrics "))
+ Expect(lyrics[0].Line[0].Cue[1].End).To(Equal(ptr(int64(2000))))
+ Expect(lyrics[0].Line[0].Cue[1].ByteStart).To(Equal(5))
+ Expect(lyrics[0].Line[0].Cue[1].ByteEnd).To(Equal(11))
+ Expect(*lyrics[0].Line[0].Cue[2].Start).To(Equal(int64(2000)))
+ Expect(lyrics[0].Line[0].Cue[2].Value).To(Equal("here"))
+ Expect(lyrics[0].Line[0].Cue[2].End).To(Equal(ptr(int64(3000))))
+ Expect(lyrics[0].Line[0].Cue[2].ByteStart).To(Equal(12))
+ Expect(lyrics[0].Line[0].Cue[2].ByteEnd).To(Equal(15))
+
+ // Line 2: has inline markers
+ Expect(lyrics[0].Line[1].Start).To(Equal(ptr(int64(3000))))
+ Expect(lyrics[0].Line[1].End).To(Equal(ptr(int64(5000))))
+ Expect(lyrics[0].Line[1].Value).To(Equal("More words"))
+ Expect(lyrics[0].Line[1].Cue).To(HaveLen(2))
+ Expect(lyrics[0].Line[1].Cue[0].End).To(Equal(ptr(int64(3500))))
+ Expect(lyrics[0].Line[1].Cue[1].End).To(Equal(ptr(int64(5000))))
+ Expect(lyrics[0].Line[1].Cue[0].ByteStart).To(Equal(0))
+ Expect(lyrics[0].Line[1].Cue[0].ByteEnd).To(Equal(4))
+ Expect(lyrics[0].Line[1].Cue[1].ByteStart).To(Equal(5))
+ Expect(lyrics[0].Line[1].Cue[1].ByteEnd).To(Equal(9))
+
+ // Line 3: plain line, no cues
+ Expect(lyrics[0].Line[2].Start).To(Equal(ptr(int64(5000))))
+ Expect(lyrics[0].Line[2].Value).To(Equal("Plain line without inline markers"))
+ Expect(lyrics[0].Line[2].Cue).To(BeNil())
+ })
+
+ It("should return Enhanced LRC lyrics from an ELRC file", func() {
+ mf := model.MediaFile{Path: "tests/fixtures/test.mp3"}
+ lyrics, err := fromExternalFile(ctx, &mf, ".elrc")
+
+ Expect(err).To(BeNil())
+ Expect(lyrics).To(HaveLen(1))
+ Expect(lyrics[0].DisplayArtist).To(Equal("ELRC Artist"))
+ Expect(lyrics[0].DisplayTitle).To(Equal("ELRC Song"))
+ Expect(lyrics[0].Lang).To(Equal("eng"))
+ Expect(lyrics[0].Synced).To(BeTrue())
+ Expect(lyrics[0].Line).To(HaveLen(2))
+
+ Expect(lyrics[0].Line[0].Start).To(Equal(ptr(int64(1000))))
+ Expect(lyrics[0].Line[0].End).To(Equal(ptr(int64(3000))))
+ Expect(lyrics[0].Line[0].Value).To(Equal("Lead words"))
+ Expect(lyrics[0].Line[0].Cue).To(HaveLen(2))
+ Expect(*lyrics[0].Line[0].Cue[0].Start).To(Equal(int64(1000)))
+ Expect(lyrics[0].Line[0].Cue[0].Value).To(Equal("Lead "))
+ Expect(lyrics[0].Line[0].Cue[0].End).To(Equal(ptr(int64(1500))))
+ Expect(lyrics[0].Line[0].Cue[0].ByteStart).To(Equal(0))
+ Expect(lyrics[0].Line[0].Cue[0].ByteEnd).To(Equal(4))
+ Expect(*lyrics[0].Line[0].Cue[1].Start).To(Equal(int64(1500)))
+ Expect(lyrics[0].Line[0].Cue[1].Value).To(Equal("words"))
+ Expect(lyrics[0].Line[0].Cue[1].End).To(Equal(ptr(int64(3000))))
+ Expect(lyrics[0].Line[0].Cue[1].ByteStart).To(Equal(5))
+ Expect(lyrics[0].Line[0].Cue[1].ByteEnd).To(Equal(9))
+
+ Expect(lyrics[0].Line[1].Start).To(Equal(ptr(int64(3000))))
+ Expect(lyrics[0].Line[1].Value).To(Equal("Fallback line"))
+ Expect(lyrics[0].Line[1].Cue).To(BeNil())
+ })
+
It("should return unsynchronized lyrics from a file", func() {
mf := model.MediaFile{Path: "tests/fixtures/test.mp3"}
lyrics, err := fromExternalFile(ctx, &mf, ".txt")
@@ -108,6 +191,66 @@ var _ = Describe("sources", func() {
}))
})
+ It("should return synchronized lyrics from an SRT file", func() {
+ mf := model.MediaFile{Path: "tests/fixtures/test.mp3"}
+ lyrics, err := fromExternalFile(ctx, &mf, ".srt")
+
+ Expect(err).To(BeNil())
+ Expect(lyrics).To(Equal(model.LyricList{
+ model.Lyrics{
+ Lang: "xxx",
+ Line: []model.Line{
+ {
+ Start: ptr(int64(18800)),
+ End: ptr(int64(22800)),
+ Value: "We're from subtitles",
+ },
+ {
+ Start: ptr(int64(22801)),
+ End: ptr(int64(26000)),
+ Value: "Another subtitle line",
+ },
+ },
+ Synced: true,
+ },
+ }))
+ })
+
+ It("should return synchronized multilingual lyrics from a TTML file", func() {
+ mf := model.MediaFile{Path: "tests/fixtures/test.mp3"}
+ lyrics, err := fromExternalFile(ctx, &mf, ".ttml")
+
+ Expect(err).To(BeNil())
+ Expect(lyrics).To(Equal(model.LyricList{
+ {
+ Kind: "main",
+ Lang: "eng",
+ Line: []model.Line{
+ {
+ Start: ptr(int64(18800)),
+ Value: "We're no strangers to love",
+ },
+ {
+ Start: ptr(int64(22800)),
+ Value: "You know the rules and so do I",
+ },
+ },
+ Synced: true,
+ },
+ {
+ Kind: "main",
+ Lang: "por",
+ Line: []model.Line{
+ {
+ Start: ptr(int64(18800)),
+ Value: "Nao somos estranhos ao amor",
+ },
+ },
+ Synced: true,
+ },
+ }))
+ })
+
It("should handle LRC files with UTF-8 BOM marker (issue #4631)", func() {
// The function looks for , so we need to pass
// a MediaFile with .mp3 path and look for .lrc suffix
@@ -141,5 +284,33 @@ var _ = Describe("sources", func() {
Expect(lyrics[0].Line[1].Start).To(Equal(new(int64(22801))))
Expect(lyrics[0].Line[1].Value).To(Equal("You know the rules and so do I"))
})
+
+ It("should handle TTML files with UTF-8 BOM marker", func() {
+ mf := model.MediaFile{Path: "tests/fixtures/bom-test.mp3"}
+ lyrics, err := fromExternalFile(ctx, &mf, ".ttml")
+
+ Expect(err).To(BeNil())
+ Expect(lyrics).To(HaveLen(1))
+ Expect(lyrics[0].Kind).To(Equal("main"))
+ Expect(lyrics[0].Synced).To(BeTrue())
+ Expect(lyrics[0].Line).To(HaveLen(1))
+ Expect(lyrics[0].Line[0].Start).To(Equal(ptr(int64(0))))
+ Expect(lyrics[0].Line[0].Value).To(Equal("BOM test line"))
+ })
+
+ It("should handle UTF-16 BE encoded TTML files", func() {
+ mf := model.MediaFile{Path: "tests/fixtures/bom-utf16-test.mp3"}
+ lyrics, err := fromExternalFile(ctx, &mf, ".ttml")
+
+ Expect(err).To(BeNil())
+ Expect(lyrics).To(HaveLen(1))
+ Expect(lyrics[0].Kind).To(Equal("main"))
+ Expect(lyrics[0].Synced).To(BeTrue())
+ Expect(lyrics[0].Line).To(HaveLen(2))
+ Expect(lyrics[0].Line[0].Start).To(Equal(ptr(int64(18800))))
+ Expect(lyrics[0].Line[0].Value).To(Equal("UTF16 line one"))
+ Expect(lyrics[0].Line[1].Start).To(Equal(ptr(int64(22801))))
+ Expect(lyrics[0].Line[1].Value).To(Equal("UTF16 line two"))
+ })
})
})
diff --git a/core/lyrics/srt.go b/core/lyrics/srt.go
new file mode 100644
index 000000000..05a43ca57
--- /dev/null
+++ b/core/lyrics/srt.go
@@ -0,0 +1,168 @@
+package lyrics
+
+import (
+ "bytes"
+ "regexp"
+ "strconv"
+ "strings"
+
+ "github.com/navidrome/navidrome/model"
+ "github.com/navidrome/navidrome/utils/str"
+)
+
+var (
+ srtTimeRegex = regexp.MustCompile(`^\s*(\d{1,2}):(\d{2}):(\d{2})[,.](\d{1,3})\s*$`)
+ srtBlockSeparatorRegex = regexp.MustCompile(`\n\s*\n`)
+)
+
+func parseSRT(contents []byte) (model.LyricList, error) {
+ return parseSRTWithLanguage(contents, "xxx")
+}
+
+func parseSRTWithLanguage(contents []byte, language string) (model.LyricList, error) {
+ raw := strings.ReplaceAll(string(contents), "\r\n", "\n")
+ raw = strings.ReplaceAll(raw, "\r", "\n")
+
+ blocks := splitSRTBlocks(raw)
+ lines := make([]model.Line, 0, len(blocks))
+
+ for _, block := range blocks {
+ line, ok, err := parseSRTBlock(block)
+ if err != nil {
+ return nil, err
+ }
+ if ok {
+ lines = append(lines, line)
+ }
+ }
+
+ if len(lines) == 0 {
+ return nil, nil
+ }
+
+ lyrics := model.NormalizeLyrics(model.Lyrics{
+ Lang: normalizeEmbeddedLanguage(language),
+ Line: lines,
+ Synced: true,
+ })
+ return model.LyricList{lyrics}, nil
+}
+
+func splitSRTBlocks(raw string) []string {
+ raw = strings.TrimSpace(raw)
+ if raw == "" {
+ return nil
+ }
+
+ parts := srtBlockSeparatorRegex.Split(raw, -1)
+ blocks := make([]string, 0, len(parts))
+ for _, part := range parts {
+ part = strings.TrimSpace(part)
+ if part != "" {
+ blocks = append(blocks, part)
+ }
+ }
+ return blocks
+}
+
+func parseSRTBlock(block string) (model.Line, bool, error) {
+ scanner := bytes.Split([]byte(block), []byte("\n"))
+ if len(scanner) == 0 {
+ return model.Line{}, false, nil
+ }
+
+ lines := make([]string, 0, len(scanner))
+ for _, line := range scanner {
+ lines = append(lines, strings.TrimSpace(string(line)))
+ }
+
+ if len(lines) == 0 {
+ return model.Line{}, false, nil
+ }
+
+ startIdx := 0
+ if digitsOnly(lines[0]) {
+ startIdx = 1
+ }
+ if startIdx >= len(lines) {
+ return model.Line{}, false, nil
+ }
+
+ timing := strings.Split(lines[startIdx], "-->")
+ if len(timing) != 2 {
+ return model.Line{}, false, nil
+ }
+
+ startMs, err := parseSRTTime(timing[0])
+ if err != nil {
+ return model.Line{}, false, err
+ }
+ endMs, err := parseSRTTime(timing[1])
+ if err != nil {
+ return model.Line{}, false, err
+ }
+
+ textLines := make([]string, 0, len(lines)-startIdx-1)
+ for _, line := range lines[startIdx+1:] {
+ if line == "" {
+ continue
+ }
+ textLines = append(textLines, line)
+ }
+
+ value := str.SanitizeText(strings.Join(textLines, "\n"))
+ if value == "" {
+ return model.Line{}, false, nil
+ }
+
+ return model.Line{
+ Start: &startMs,
+ End: &endMs,
+ Value: value,
+ }, true, nil
+}
+
+func parseSRTTime(value string) (int64, error) {
+ match := srtTimeRegex.FindStringSubmatch(strings.TrimSpace(value))
+ if match == nil {
+ return 0, strconv.ErrSyntax
+ }
+
+ hours, err := strconv.ParseInt(match[1], 10, 64)
+ if err != nil {
+ return 0, err
+ }
+ minutes, err := strconv.ParseInt(match[2], 10, 64)
+ if err != nil {
+ return 0, err
+ }
+ seconds, err := strconv.ParseInt(match[3], 10, 64)
+ if err != nil {
+ return 0, err
+ }
+ millis, err := strconv.ParseInt(match[4], 10, 64)
+ if err != nil {
+ return 0, err
+ }
+
+ switch len(match[4]) {
+ case 1:
+ millis *= 100
+ case 2:
+ millis *= 10
+ }
+
+ return (((hours*60)+minutes)*60+seconds)*1000 + millis, nil
+}
+
+func digitsOnly(value string) bool {
+ if value == "" {
+ return false
+ }
+ for _, ch := range value {
+ if ch < '0' || ch > '9' {
+ return false
+ }
+ }
+ return true
+}
diff --git a/core/lyrics/test_helpers_test.go b/core/lyrics/test_helpers_test.go
new file mode 100644
index 000000000..a4a03f0fe
--- /dev/null
+++ b/core/lyrics/test_helpers_test.go
@@ -0,0 +1,5 @@
+package lyrics
+
+func ptr[T any](v T) *T {
+ return &v
+}
diff --git a/core/lyrics/ttml.go b/core/lyrics/ttml.go
new file mode 100644
index 000000000..163bd657c
--- /dev/null
+++ b/core/lyrics/ttml.go
@@ -0,0 +1,1278 @@
+package lyrics
+
+import (
+ "bytes"
+ "encoding/xml"
+ "errors"
+ "io"
+ "math"
+ "regexp"
+ "sort"
+ "strconv"
+ "strings"
+ "unicode"
+
+ "github.com/navidrome/navidrome/log"
+ "github.com/navidrome/navidrome/model"
+ "github.com/navidrome/navidrome/utils/str"
+)
+
+const (
+ defaultTTMLFrameRate = 30.0
+ defaultTTMLSubFrameRate = 1.0
+ defaultTTMLTickRate = 1.0
+
+ ttmlLyricKindMain = "main"
+ ttmlLyricKindTranslation = "translation"
+ ttmlLyricKindPronunciation = "pronunciation"
+ ttmlBackgroundAgentPrefix = "__nd_bg__|"
+)
+
+var offsetTimeRegex = regexp.MustCompile(`^([0-9]+(?:\.[0-9]+)?)(h|m|s|ms|f|t)$`)
+var xmlEncodingRegex = regexp.MustCompile(`(?i)<\?xml([^>]*?)encoding\s*=\s*["'][^"']+["']([^>]*)\?>`)
+
+type ttmlTimeKind int
+
+const (
+ ttmlTimeAbsolute ttmlTimeKind = iota
+ ttmlTimeOffset
+ ttmlTimeAmbiguous
+)
+
+type ttmlTimingParams struct {
+ frameRate float64
+ subFrameRate float64
+ tickRate float64
+}
+
+type ttmlTimingContext struct {
+ lang string
+ role string
+ agentID string
+ begin int64
+ hasBegin bool
+ end int64
+ hasEnd bool
+ invalid bool
+}
+
+type ttmlLineRef struct {
+ order int
+ line model.Line
+}
+
+type ttmlMetadataEntry struct {
+ key string
+ line model.Line
+ seq int
+}
+
+type ttmlResolvedMetadataLine struct {
+ order int
+ seq int
+ line model.Line
+}
+
+type ttmlDefinedAgent struct {
+ ID string
+ Type string
+ Name string
+}
+
+type ttmlPiece struct {
+ raw string
+ cue *model.Cue
+}
+
+type ttmlParser struct {
+ decoder *xml.Decoder
+ params ttmlTimingParams
+
+ mainLangOrder []string
+ mainLinesByLang map[string][]model.Line
+
+ mainLineRefsByKey map[string]ttmlLineRef
+ mainLineOrder int
+
+ translationLangOrder []string
+ translationEntriesByLg map[string][]ttmlMetadataEntry
+
+ pronunciationLangOrder []string
+ pronunciationEntriesByLg map[string][]ttmlMetadataEntry
+
+ definedAgents map[string]ttmlDefinedAgent
+
+ metadataSeq int
+}
+
+func parseTTML(contents []byte) (model.LyricList, error) {
+ return parseTTMLWithDefaultLang(contents, "xxx")
+}
+
+func parseTTMLWithDefaultLang(contents []byte, defaultLang string) (model.LyricList, error) {
+ contents = xmlEncodingRegex.ReplaceAll(contents, []byte(``))
+
+ p := ttmlParser{
+ decoder: xml.NewDecoder(bytes.NewReader(contents)),
+ params: ttmlTimingParams{
+ frameRate: defaultTTMLFrameRate,
+ subFrameRate: defaultTTMLSubFrameRate,
+ tickRate: defaultTTMLTickRate,
+ },
+ mainLinesByLang: make(map[string][]model.Line),
+ mainLineRefsByKey: make(map[string]ttmlLineRef),
+ translationEntriesByLg: make(map[string][]ttmlMetadataEntry),
+ pronunciationEntriesByLg: make(map[string][]ttmlMetadataEntry),
+ definedAgents: make(map[string]ttmlDefinedAgent),
+ }
+
+ root := ttmlTimingContext{lang: normalizeTTMLLang(defaultLang)}
+
+ for {
+ token, err := p.decoder.Token()
+ if errors.Is(err, io.EOF) {
+ break
+ }
+ if err != nil {
+ return nil, err
+ }
+
+ start, ok := token.(xml.StartElement)
+ if !ok {
+ continue
+ }
+
+ if err := p.parseElement(start, root); err != nil {
+ return nil, err
+ }
+ }
+
+ return p.toLyricList(), nil
+}
+
+func (p *ttmlParser) parseElement(start xml.StartElement, parent ttmlTimingContext) error {
+ local := strings.ToLower(start.Name.Local)
+ if local == "tt" {
+ p.updateTimingParams(start.Attr)
+ }
+
+ switch local {
+ case "translation":
+ return p.parseMetadataTrack(start, parent, ttmlLyricKindTranslation)
+ case "transliteration":
+ return p.parseMetadataTrack(start, parent, ttmlLyricKindPronunciation)
+ case "agent":
+ return p.parseAgentDefinition(start)
+ }
+
+ ctx := p.childContext(start.Attr, parent)
+ if local == "p" {
+ lineText, tokens, err := p.parseParagraph(ctx)
+ if err != nil {
+ return err
+ }
+ if ctx.invalid || lineText == "" {
+ return nil
+ }
+
+ parsedLine := model.Line{Value: lineText}
+ if ctx.hasBegin {
+ startMs := ctx.begin
+ parsedLine.Start = &startMs
+ }
+ if ctx.hasEnd {
+ endMs := ctx.end
+ parsedLine.End = &endMs
+ }
+ if len(tokens) > 0 {
+ parsedLine.Cue = tokens
+ }
+ parsedLine = hydrateLineTimingFromTokens(parsedLine)
+
+ lineKey, _ := attrValue(start.Attr, "key")
+ p.addMainLine(ctx.lang, lineKey, parsedLine)
+ return nil
+ }
+
+ for {
+ token, err := p.decoder.Token()
+ if err != nil {
+ return err
+ }
+
+ switch t := token.(type) {
+ case xml.StartElement:
+ nextParent := ctx
+ if ctx.invalid {
+ // Best effort: ignore invalid timing in container elements, and
+ // continue traversing descendants with parent context.
+ nextParent = parent
+ }
+ if err := p.parseElement(t, nextParent); err != nil {
+ return err
+ }
+ case xml.EndElement:
+ if strings.EqualFold(t.Name.Local, start.Name.Local) {
+ return nil
+ }
+ }
+ }
+}
+
+func (p *ttmlParser) parseMetadataTrack(start xml.StartElement, parent ttmlTimingContext, kind string) error {
+ ctx := p.childContext(start.Attr, parent)
+ lang := normalizeTTMLLang(ctx.lang)
+
+ for {
+ token, err := p.decoder.Token()
+ if err != nil {
+ return err
+ }
+
+ switch t := token.(type) {
+ case xml.StartElement:
+ if strings.EqualFold(t.Name.Local, "text") {
+ entry, ok, err := p.parseMetadataText(t, ctx)
+ if err != nil {
+ return err
+ }
+ if ok {
+ p.addMetadataEntry(kind, lang, entry)
+ }
+ continue
+ }
+
+ nextParent := ctx
+ if ctx.invalid {
+ nextParent = parent
+ }
+ if err := p.parseElement(t, nextParent); err != nil {
+ return err
+ }
+ case xml.EndElement:
+ if strings.EqualFold(t.Name.Local, start.Name.Local) {
+ return nil
+ }
+ }
+ }
+}
+
+func (p *ttmlParser) parseAgentDefinition(start xml.StartElement) error {
+ id, ok := attrValue(start.Attr, "id")
+ id = strings.TrimSpace(id)
+ if !ok || id == "" {
+ return p.skipElement(start)
+ }
+
+ agent := ttmlDefinedAgent{
+ ID: id,
+ Type: strings.ToLower(strings.TrimSpace(attrOrEmpty(start.Attr, "type"))),
+ }
+
+ for {
+ token, err := p.decoder.Token()
+ if err != nil {
+ return err
+ }
+
+ switch t := token.(type) {
+ case xml.StartElement:
+ if strings.EqualFold(t.Name.Local, "name") {
+ name, err := p.collectElementText(t)
+ if err != nil {
+ return err
+ }
+ name = sanitizeTTMLText(name)
+ if name != "" && agent.Name == "" {
+ agent.Name = name
+ }
+ continue
+ }
+ if err := p.skipElement(t); err != nil {
+ return err
+ }
+ case xml.EndElement:
+ if strings.EqualFold(t.Name.Local, start.Name.Local) {
+ p.definedAgents[agent.ID] = agent
+ return nil
+ }
+ }
+ }
+}
+
+func (p *ttmlParser) parseMetadataText(start xml.StartElement, parent ttmlTimingContext) (ttmlMetadataEntry, bool, error) {
+ forKey, hasFor := attrValue(start.Attr, "for")
+ forKey = strings.TrimSpace(forKey)
+
+ pieces, err := p.parseInlineElement(start, parent)
+ if err != nil {
+ return ttmlMetadataEntry{}, false, err
+ }
+ if !hasFor || forKey == "" {
+ return ttmlMetadataEntry{}, false, nil
+ }
+
+ ctx := p.childContext(start.Attr, parent)
+ if ctx.invalid {
+ return ttmlMetadataEntry{}, false, nil
+ }
+
+ value, tokens := buildTTMLLineFromPieces(pieces)
+ line := model.Line{Value: value}
+ if ctx.hasBegin {
+ startMs := ctx.begin
+ line.Start = &startMs
+ }
+ if ctx.hasEnd {
+ endMs := ctx.end
+ line.End = &endMs
+ }
+ if len(tokens) > 0 {
+ line.Cue = tokens
+ }
+ line = hydrateLineTimingFromTokens(line)
+
+ if line.Value == "" && len(line.Cue) == 0 {
+ return ttmlMetadataEntry{}, false, nil
+ }
+
+ return ttmlMetadataEntry{key: forKey, line: line}, true, nil
+}
+
+func (p *ttmlParser) parseParagraph(parent ttmlTimingContext) (string, []model.Cue, error) {
+ var pieces []ttmlPiece
+
+ for {
+ token, err := p.decoder.Token()
+ if err != nil {
+ return "", nil, err
+ }
+
+ switch t := token.(type) {
+ case xml.StartElement:
+ inlinePieces, err := p.parseInlineElement(t, parent)
+ if err != nil {
+ return "", nil, err
+ }
+ pieces = append(pieces, inlinePieces...)
+ case xml.EndElement:
+ if strings.EqualFold(t.Name.Local, "p") {
+ value, tokens := buildTTMLLineFromPieces(pieces)
+ return value, tokens, nil
+ }
+ case xml.CharData:
+ pieces = append(pieces, ttmlPiece{raw: string(t)})
+ }
+ }
+}
+
+func (p *ttmlParser) parseInlineElement(start xml.StartElement, parent ttmlTimingContext) ([]ttmlPiece, error) {
+ local := strings.ToLower(start.Name.Local)
+ if local == "br" {
+ return []ttmlPiece{{raw: "\n"}}, nil
+ }
+
+ ctx := p.childContext(start.Attr, parent)
+ _, hasBegin := attrValue(start.Attr, "begin")
+ _, hasEnd := attrValue(start.Attr, "end")
+ _, hasDur := attrValue(start.Attr, "dur")
+ hasOwnTiming := hasBegin || hasEnd || hasDur
+
+ var pieces []ttmlPiece
+
+ for {
+ token, err := p.decoder.Token()
+ if err != nil {
+ return nil, err
+ }
+
+ switch t := token.(type) {
+ case xml.StartElement:
+ inlinePieces, err := p.parseInlineElement(t, ctx)
+ if err != nil {
+ return nil, err
+ }
+ pieces = append(pieces, inlinePieces...)
+ case xml.EndElement:
+ if !strings.EqualFold(t.Name.Local, start.Name.Local) {
+ continue
+ }
+
+ if local == "span" && hasOwnTiming && !ctx.invalid && !ttmlPiecesContainCue(pieces) {
+ rawValue := concatTTMLPieceRaw(pieces)
+ tokenText := sanitizeTTMLText(rawValue)
+ if tokenText != "" {
+ parsedToken := model.Cue{
+ AgentID: p.resolveCueAgentID(ctx),
+ }
+ if ctx.hasBegin {
+ startMs := ctx.begin
+ parsedToken.Start = &startMs
+ }
+ if ctx.hasEnd {
+ endMs := ctx.end
+ parsedToken.End = &endMs
+ }
+
+ return []ttmlPiece{{
+ raw: rawValue,
+ cue: &parsedToken,
+ }}, nil
+ }
+ }
+
+ return pieces, nil
+ case xml.CharData:
+ pieces = append(pieces, ttmlPiece{raw: string(t)})
+ }
+ }
+}
+
+func buildTTMLLineFromPieces(pieces []ttmlPiece) (string, []model.Cue) {
+ finalized := finalizeTTMLLines(splitTTMLPiecesByNewline(pieces))
+ for len(finalized) > 0 && finalized[0].text == "" && len(finalized[0].cues) == 0 {
+ finalized = finalized[1:]
+ }
+ for len(finalized) > 0 {
+ last := finalized[len(finalized)-1]
+ if last.text != "" || len(last.cues) > 0 {
+ break
+ }
+ finalized = finalized[:len(finalized)-1]
+ }
+
+ var value strings.Builder
+ cues := make([]model.Cue, 0, 8)
+ byteOffset := 0
+ for i, line := range finalized {
+ if i > 0 {
+ value.WriteByte('\n')
+ byteOffset++
+ }
+ value.WriteString(line.text)
+ for _, cue := range line.cues {
+ cue.ByteStart += byteOffset
+ cue.ByteEnd += byteOffset
+ cues = append(cues, cue)
+ }
+ byteOffset += len(line.text)
+ }
+
+ return value.String(), cues
+}
+
+type ttmlFinalLine struct {
+ text string
+ cues []model.Cue
+}
+
+func finalizeTTMLLines(lines [][]ttmlPiece) []ttmlFinalLine {
+ finalized := make([]ttmlFinalLine, 0, len(lines))
+ for _, line := range lines {
+ text, cues := finalizeTTMLLogicalLine(line)
+ finalized = append(finalized, ttmlFinalLine{text: text, cues: cues})
+ }
+ return finalized
+}
+
+func splitTTMLPiecesByNewline(pieces []ttmlPiece) [][]ttmlPiece {
+ lines := [][]ttmlPiece{{}}
+ for _, piece := range pieces {
+ raw := normalizeTTMLPieceRaw(piece.raw)
+ if raw == "" {
+ continue
+ }
+
+ start := 0
+ for i := 0; i < len(raw); i++ {
+ if raw[i] != '\n' {
+ continue
+ }
+ if start < i {
+ lines[len(lines)-1] = append(lines[len(lines)-1], ttmlPiece{
+ raw: raw[start:i],
+ cue: cloneTTMLCue(piece.cue),
+ })
+ }
+ lines = append(lines, []ttmlPiece{})
+ start = i + 1
+ }
+ if start < len(raw) {
+ lines[len(lines)-1] = append(lines[len(lines)-1], ttmlPiece{
+ raw: raw[start:],
+ cue: cloneTTMLCue(piece.cue),
+ })
+ }
+ }
+ return lines
+}
+
+func finalizeTTMLLogicalLine(line []ttmlPiece) (string, []model.Cue) {
+ rawLine := concatTTMLPieceRaw(line)
+ if rawLine == "" {
+ return "", nil
+ }
+
+ leftTrimBytes := len(rawLine) - len(strings.TrimLeftFunc(rawLine, unicode.IsSpace))
+ rightTrimBytes := len(rawLine) - len(strings.TrimRightFunc(rawLine, unicode.IsSpace))
+ trimmedEnd := len(rawLine) - rightTrimBytes
+ if trimmedEnd < leftTrimBytes {
+ trimmedEnd = leftTrimBytes
+ }
+
+ trimmed := strings.TrimSpace(rawLine)
+ cues := make([]model.Cue, 0, len(line))
+ cursor := 0
+ for _, piece := range line {
+ pieceEnd := cursor + len(piece.raw)
+ if piece.cue != nil {
+ byteStart := max(cursor, leftTrimBytes)
+ byteEnd := min(pieceEnd, trimmedEnd)
+ if byteStart < byteEnd {
+ cue := *piece.cue
+ cue.Value = rawLine[byteStart:byteEnd]
+ cue.ByteStart = byteStart - leftTrimBytes
+ cue.ByteEnd = byteEnd - leftTrimBytes - 1
+ cues = append(cues, cue)
+ }
+ }
+ cursor = pieceEnd
+ }
+
+ return trimmed, cues
+}
+
+func normalizeTTMLPieceRaw(raw string) string {
+ raw = str.SanitizeText(raw)
+ raw = strings.ReplaceAll(raw, "\r\n", "\n")
+ raw = strings.ReplaceAll(raw, "\r", "\n")
+ return raw
+}
+
+func concatTTMLPieceRaw(pieces []ttmlPiece) string {
+ var raw strings.Builder
+ for _, piece := range pieces {
+ raw.WriteString(normalizeTTMLPieceRaw(piece.raw))
+ }
+ return raw.String()
+}
+
+func ttmlPiecesContainCue(pieces []ttmlPiece) bool {
+ for _, piece := range pieces {
+ if piece.cue != nil {
+ return true
+ }
+ }
+ return false
+}
+
+func cloneTTMLCue(cue *model.Cue) *model.Cue {
+ if cue == nil {
+ return nil
+ }
+
+ cloned := *cue
+ return &cloned
+}
+
+func (p *ttmlParser) toLyricList() model.LyricList {
+ res := make(model.LyricList, 0, len(p.mainLangOrder)+len(p.translationLangOrder)+len(p.pronunciationLangOrder))
+ for _, lang := range p.mainLangOrder {
+ lines := p.mainLinesByLang[lang]
+ if len(lines) == 0 {
+ continue
+ }
+ res = append(res, p.finalizeLyrics(model.Lyrics{
+ Kind: ttmlLyricKindMain,
+ Lang: lang,
+ Line: lines,
+ Synced: linesAreSynced(lines),
+ }))
+ }
+
+ res = append(res, p.buildMetadataLyrics(ttmlLyricKindTranslation, p.translationLangOrder, p.translationEntriesByLg)...)
+ res = append(res, p.buildMetadataLyrics(ttmlLyricKindPronunciation, p.pronunciationLangOrder, p.pronunciationEntriesByLg)...)
+ return res
+}
+
+func (p *ttmlParser) buildMetadataLyrics(kind string, langOrder []string, entriesByLang map[string][]ttmlMetadataEntry) model.LyricList {
+ res := make(model.LyricList, 0, len(langOrder))
+
+ for _, lang := range langOrder {
+ entries := entriesByLang[lang]
+ if len(entries) == 0 {
+ continue
+ }
+
+ seenKeys := make(map[string]struct{}, len(entries))
+ resolved := make([]ttmlResolvedMetadataLine, 0, len(entries))
+ for _, entry := range entries {
+ if _, exists := seenKeys[entry.key]; exists {
+ continue
+ }
+ seenKeys[entry.key] = struct{}{}
+
+ ref, ok := p.mainLineRefsByKey[entry.key]
+ if !ok {
+ log.Warn("Skipping TTML metadata line without matching key", "kind", kind, "lang", lang, "key", entry.key)
+ continue
+ }
+
+ line := entry.line
+ if line.Start == nil && ref.line.Start != nil {
+ startMs := *ref.line.Start
+ line.Start = &startMs
+ }
+ if line.End == nil && ref.line.End != nil {
+ endMs := *ref.line.End
+ line.End = &endMs
+ }
+ line = hydrateLineTimingFromTokens(line)
+
+ if line.Value == "" && len(line.Cue) == 0 {
+ continue
+ }
+
+ resolved = append(resolved, ttmlResolvedMetadataLine{
+ order: ref.order,
+ seq: entry.seq,
+ line: line,
+ })
+ }
+
+ if len(resolved) == 0 {
+ continue
+ }
+
+ sort.SliceStable(resolved, func(i, j int) bool {
+ if resolved[i].order != resolved[j].order {
+ return resolved[i].order < resolved[j].order
+ }
+ return resolved[i].seq < resolved[j].seq
+ })
+
+ lines := make([]model.Line, len(resolved))
+ for i := range resolved {
+ lines[i] = resolved[i].line
+ }
+
+ res = append(res, p.finalizeLyrics(model.Lyrics{
+ Kind: kind,
+ Lang: lang,
+ Line: lines,
+ Synced: linesAreSynced(lines),
+ }))
+ }
+
+ return res
+}
+
+func (p *ttmlParser) finalizeLyrics(lyrics model.Lyrics) model.Lyrics {
+ lyrics.Line, lyrics.Agents = p.resolveAgents(lyrics.Line)
+ return model.NormalizeLyrics(lyrics)
+}
+
+func (p *ttmlParser) resolveAgents(lines []model.Line) ([]model.Line, []model.Agent) {
+ if len(lines) == 0 {
+ return lines, nil
+ }
+
+ usedOrder := make([]string, 0, 4)
+ usedSet := make(map[string]struct{}, 4)
+ sawEmptyCue := false
+
+ for i := range lines {
+ for j := range lines[i].Cue {
+ agentID := strings.TrimSpace(lines[i].Cue[j].AgentID)
+ if agentID == "" {
+ sawEmptyCue = true
+ continue
+ }
+ if _, exists := usedSet[agentID]; !exists {
+ usedSet[agentID] = struct{}{}
+ usedOrder = append(usedOrder, agentID)
+ }
+ }
+ }
+
+ if len(usedOrder) == 0 {
+ return lines, nil
+ }
+
+ mainID := ""
+ for _, agentID := range usedOrder {
+ role := p.baseRoleForAgent(agentID)
+ if role != "bg" && role != "group" {
+ mainID = agentID
+ break
+ }
+ }
+ if mainID == "" && sawEmptyCue {
+ mainID = "main"
+ }
+ if mainID == "" {
+ for _, agentID := range usedOrder {
+ if p.baseRoleForAgent(agentID) != "bg" {
+ mainID = agentID
+ break
+ }
+ }
+ }
+ if mainID == "" {
+ mainID = usedOrder[0]
+ }
+
+ if _, exists := usedSet[mainID]; !exists {
+ usedSet[mainID] = struct{}{}
+ usedOrder = append([]string{mainID}, usedOrder...)
+ }
+
+ for i := range lines {
+ for j := range lines[i].Cue {
+ if strings.TrimSpace(lines[i].Cue[j].AgentID) == "" {
+ lines[i].Cue[j].AgentID = mainID
+ }
+ }
+ }
+
+ agents := make([]model.Agent, 0, len(usedOrder))
+ for _, agentID := range usedOrder {
+ role := p.baseRoleForAgent(agentID)
+ if agentID == mainID {
+ role = "main"
+ }
+ agent := model.Agent{
+ ID: agentID,
+ Role: role,
+ Name: p.agentNameForID(agentID),
+ }
+ agents = append(agents, agent)
+ }
+
+ return lines, agents
+}
+
+func (p *ttmlParser) resolveCueAgentID(ctx ttmlTimingContext) string {
+ agentID := strings.TrimSpace(ctx.agentID)
+ if contextHasRole(ctx.role, "x-bg") {
+ if agentID == "" {
+ agentID = "main"
+ }
+ return backgroundAgentID(agentID)
+ }
+ return agentID
+}
+
+func (p *ttmlParser) baseRoleForAgent(agentID string) string {
+ if isBackgroundAgentID(agentID) {
+ return "bg"
+ }
+
+ if agent, ok := p.definedAgents[agentID]; ok {
+ switch agent.Type {
+ case "group":
+ return "group"
+ default:
+ return "voice"
+ }
+ }
+
+ return "voice"
+}
+
+func (p *ttmlParser) agentNameForID(agentID string) string {
+ if isBackgroundAgentID(agentID) {
+ baseID := strings.TrimPrefix(agentID, ttmlBackgroundAgentPrefix)
+ if baseID == "main" {
+ return ""
+ }
+ if agent, ok := p.definedAgents[baseID]; ok {
+ return agent.Name
+ }
+ return ""
+ }
+
+ if agent, ok := p.definedAgents[agentID]; ok {
+ return agent.Name
+ }
+
+ return ""
+}
+
+func backgroundAgentID(agentID string) string {
+ return ttmlBackgroundAgentPrefix + agentID
+}
+
+func isBackgroundAgentID(agentID string) bool {
+ return strings.HasPrefix(agentID, ttmlBackgroundAgentPrefix)
+}
+
+func contextHasRole(roles string, role string) bool {
+ for _, candidate := range strings.Fields(strings.ToLower(roles)) {
+ if candidate == strings.ToLower(role) {
+ return true
+ }
+ }
+ return false
+}
+
+func appendTTMLRoles(existing string, roles string) string {
+ for _, role := range strings.Fields(roles) {
+ if contextHasRole(existing, role) {
+ continue
+ }
+ if existing == "" {
+ existing = role
+ } else {
+ existing += " " + role
+ }
+ }
+ return existing
+}
+
+func (p *ttmlParser) addMainLine(lang string, lineKey string, line model.Line) {
+ lang = normalizeTTMLLang(lang)
+ if _, ok := p.mainLinesByLang[lang]; !ok {
+ p.mainLangOrder = append(p.mainLangOrder, lang)
+ }
+ p.mainLinesByLang[lang] = append(p.mainLinesByLang[lang], line)
+
+ lineKey = strings.TrimSpace(lineKey)
+ if lineKey != "" {
+ if _, exists := p.mainLineRefsByKey[lineKey]; !exists {
+ p.mainLineRefsByKey[lineKey] = ttmlLineRef{
+ order: p.mainLineOrder,
+ line: line,
+ }
+ }
+ }
+ p.mainLineOrder++
+}
+
+func (p *ttmlParser) addMetadataEntry(kind string, lang string, entry ttmlMetadataEntry) {
+ lang = normalizeTTMLLang(lang)
+ entry.seq = p.metadataSeq
+ p.metadataSeq++
+
+ switch kind {
+ case ttmlLyricKindTranslation:
+ if _, ok := p.translationEntriesByLg[lang]; !ok {
+ p.translationLangOrder = append(p.translationLangOrder, lang)
+ }
+ p.translationEntriesByLg[lang] = append(p.translationEntriesByLg[lang], entry)
+ case ttmlLyricKindPronunciation:
+ if _, ok := p.pronunciationEntriesByLg[lang]; !ok {
+ p.pronunciationLangOrder = append(p.pronunciationLangOrder, lang)
+ }
+ p.pronunciationEntriesByLg[lang] = append(p.pronunciationEntriesByLg[lang], entry)
+ }
+}
+
+func (p *ttmlParser) childContext(attrs []xml.Attr, parent ttmlTimingContext) ttmlTimingContext {
+ ctx := parent
+
+ if lang, ok := attrValue(attrs, "lang"); ok {
+ ctx.lang = normalizeTTMLLang(lang)
+ }
+ if agentID, ok := attrValue(attrs, "agent"); ok {
+ ctx.agentID = strings.TrimSpace(agentID)
+ }
+ if role, ok := attrValue(attrs, "role"); ok {
+ role = strings.TrimSpace(role)
+ if role != "" {
+ ctx.role = appendTTMLRoles(ctx.role, role)
+ }
+ }
+
+ beginExpr, hasBegin := attrValue(attrs, "begin")
+ endExpr, hasEnd := attrValue(attrs, "end")
+ durExpr, hasDur := attrValue(attrs, "dur")
+
+ if hasBegin {
+ begin, kind, ok := parseTTMLTimeExpression(beginExpr, p.params)
+ if !ok {
+ ctx.invalid = true
+ return ctx
+ }
+
+ base := int64(0)
+ if parent.hasBegin {
+ base = parent.begin
+ }
+ ctx.begin = resolveTTMLTime(begin, kind, base, parent)
+ ctx.hasBegin = true
+ } else {
+ ctx.begin = parent.begin
+ ctx.hasBegin = parent.hasBegin
+ }
+
+ var calculatedEnd int64
+ calculatedHasEnd := false
+
+ if hasEnd {
+ end, kind, ok := parseTTMLTimeExpression(endExpr, p.params)
+ if !ok {
+ ctx.invalid = true
+ return ctx
+ }
+
+ base := ctx.begin
+ if !ctx.hasBegin {
+ base = parent.begin
+ }
+ calculatedEnd = resolveTTMLTime(end, kind, base, parent)
+ calculatedHasEnd = true
+ }
+
+ if hasDur {
+ dur, ok := parseTTMLDurationExpression(durExpr, p.params)
+ if !ok {
+ ctx.invalid = true
+ return ctx
+ }
+ if ctx.hasBegin {
+ durEnd := ctx.begin + dur
+ if !calculatedHasEnd || durEnd < calculatedEnd {
+ calculatedEnd = durEnd
+ calculatedHasEnd = true
+ }
+ }
+ }
+
+ if !calculatedHasEnd && parent.hasEnd {
+ calculatedEnd = parent.end
+ calculatedHasEnd = true
+ }
+
+ ctx.end = calculatedEnd
+ ctx.hasEnd = calculatedHasEnd
+ return ctx
+}
+
+func (p *ttmlParser) updateTimingParams(attrs []xml.Attr) {
+ frameRate := p.params.frameRate
+ if value, ok := attrValue(attrs, "frameRate"); ok {
+ if parsed, err := strconv.ParseFloat(value, 64); err == nil && parsed > 0 {
+ frameRate = parsed
+ }
+ }
+
+ if value, ok := attrValue(attrs, "frameRateMultiplier"); ok {
+ parts := strings.Fields(value)
+ if len(parts) == 2 {
+ numerator, errA := strconv.ParseFloat(parts[0], 64)
+ denominator, errB := strconv.ParseFloat(parts[1], 64)
+ if errA == nil && errB == nil && denominator > 0 {
+ frameRate = frameRate * (numerator / denominator)
+ }
+ }
+ }
+
+ subFrameRate := p.params.subFrameRate
+ if value, ok := attrValue(attrs, "subFrameRate"); ok {
+ if parsed, err := strconv.ParseFloat(value, 64); err == nil && parsed > 0 {
+ subFrameRate = parsed
+ }
+ }
+
+ tickRate := p.params.tickRate
+ if value, ok := attrValue(attrs, "tickRate"); ok {
+ if parsed, err := strconv.ParseFloat(value, 64); err == nil && parsed > 0 {
+ tickRate = parsed
+ }
+ }
+
+ p.params.frameRate = positiveOrDefault(frameRate, defaultTTMLFrameRate)
+ p.params.subFrameRate = positiveOrDefault(subFrameRate, defaultTTMLSubFrameRate)
+ p.params.tickRate = positiveOrDefault(tickRate, defaultTTMLTickRate)
+}
+
+func parseTTMLDurationExpression(expr string, params ttmlTimingParams) (int64, bool) {
+ value, _, ok := parseTTMLTimeExpression(expr, params)
+ return value, ok
+}
+
+func resolveTTMLTime(value int64, kind ttmlTimeKind, base int64, parent ttmlTimingContext) int64 {
+ switch kind {
+ case ttmlTimeAbsolute:
+ return value
+ case ttmlTimeOffset:
+ return base + value
+ case ttmlTimeAmbiguous:
+ absolute := value
+ offset := base + value
+
+ // No parent timing context → no reference frame for offsets.
+ // Prefer absolute when offset differs (i.e., base > 0).
+ if !parent.hasBegin && !parent.hasEnd && base != 0 {
+ return absolute
+ }
+
+ if parent.hasBegin && parent.hasEnd {
+ absoluteInParent := absolute >= parent.begin && absolute <= parent.end
+ offsetInParent := offset >= parent.begin && offset <= parent.end
+ if absoluteInParent && !offsetInParent {
+ return absolute
+ }
+ if offsetInParent && !absoluteInParent {
+ return offset
+ }
+ }
+
+ if parent.hasBegin {
+ if absolute < parent.begin && offset >= parent.begin {
+ return offset
+ }
+ if absolute >= parent.begin && offset > absolute {
+ return absolute
+ }
+ }
+ return offset
+ default:
+ return base + value
+ }
+}
+
+func parseTTMLTimeExpression(expr string, params ttmlTimingParams) (int64, ttmlTimeKind, bool) {
+ expr = strings.TrimSpace(expr)
+ if expr == "" {
+ return 0, ttmlTimeOffset, false
+ }
+
+ lower := strings.ToLower(expr)
+ if strings.Contains(lower, "wallclock(") ||
+ strings.Contains(lower, ".begin") ||
+ strings.Contains(lower, ".end") {
+ log.Warn("Unsupported TTML time expression", "value", expr)
+ return 0, ttmlTimeOffset, false
+ }
+
+ // Best-effort support for non-standard TTML seen in the wild where a
+ // bare decimal value is used (implicitly seconds), e.g. "0.170".
+ if value, err := strconv.ParseFloat(lower, 64); err == nil && value >= 0 {
+ return int64(math.Round(value * 1000)), ttmlTimeAmbiguous, true
+ }
+
+ if matches := offsetTimeRegex.FindStringSubmatch(lower); len(matches) == 3 {
+ value, err := strconv.ParseFloat(matches[1], 64)
+ if err != nil {
+ return 0, ttmlTimeOffset, false
+ }
+
+ unit := matches[2]
+ seconds := 0.0
+ switch unit {
+ case "h":
+ seconds = value * 60 * 60
+ case "m":
+ seconds = value * 60
+ case "s":
+ seconds = value
+ case "ms":
+ seconds = value / 1000
+ case "f":
+ seconds = value / params.frameRate
+ case "t":
+ seconds = value / params.tickRate
+ default:
+ return 0, ttmlTimeOffset, false
+ }
+
+ return int64(math.Round(seconds * 1000)), ttmlTimeOffset, true
+ }
+
+ colonCount := strings.Count(expr, ":")
+ switch colonCount {
+ case 1, 2:
+ clockMs, ok := parseTTMLClockTime(expr)
+ if !ok {
+ return 0, ttmlTimeAbsolute, false
+ }
+ return clockMs, ttmlTimeAbsolute, true
+ case 3:
+ framesMs, ok := parseTTMLFrameTime(expr, params)
+ if !ok {
+ return 0, ttmlTimeAbsolute, false
+ }
+ return framesMs, ttmlTimeAbsolute, true
+ default:
+ log.Warn("Unsupported TTML time expression", "value", expr)
+ return 0, ttmlTimeOffset, false
+ }
+}
+
+func parseTTMLClockTime(value string) (int64, bool) {
+ parts := strings.Split(value, ":")
+ if len(parts) != 2 && len(parts) != 3 {
+ return 0, false
+ }
+
+ hours := int64(0)
+ minutesIdx := 0
+ if len(parts) == 3 {
+ h, err := strconv.ParseInt(parts[0], 10, 64)
+ if err != nil {
+ return 0, false
+ }
+ hours = h
+ minutesIdx = 1
+ }
+
+ minutes, err := strconv.ParseInt(parts[minutesIdx], 10, 64)
+ if err != nil {
+ return 0, false
+ }
+
+ seconds, err := strconv.ParseFloat(parts[minutesIdx+1], 64)
+ if err != nil {
+ return 0, false
+ }
+
+ totalSeconds := float64(hours*60*60+minutes*60) + seconds
+ return int64(math.Round(totalSeconds * 1000)), true
+}
+
+func parseTTMLFrameTime(value string, params ttmlTimingParams) (int64, bool) {
+ parts := strings.Split(value, ":")
+ if len(parts) != 4 {
+ return 0, false
+ }
+
+ hours, err := strconv.ParseInt(parts[0], 10, 64)
+ if err != nil {
+ return 0, false
+ }
+
+ minutes, err := strconv.ParseInt(parts[1], 10, 64)
+ if err != nil {
+ return 0, false
+ }
+
+ seconds, err := strconv.ParseInt(parts[2], 10, 64)
+ if err != nil {
+ return 0, false
+ }
+
+ frameParts := strings.SplitN(parts[3], ".", 2)
+ frames, err := strconv.ParseFloat(frameParts[0], 64)
+ if err != nil {
+ return 0, false
+ }
+
+ subFrames := 0.0
+ if len(frameParts) == 2 {
+ subFrames, err = strconv.ParseFloat(frameParts[1], 64)
+ if err != nil {
+ return 0, false
+ }
+ }
+
+ totalSeconds := float64(hours*60*60 + minutes*60 + seconds)
+ totalSeconds += frames / params.frameRate
+ totalSeconds += subFrames / (params.subFrameRate * params.frameRate)
+
+ return int64(math.Round(totalSeconds * 1000)), true
+}
+
+func attrValue(attrs []xml.Attr, key string) (string, bool) {
+ for _, attr := range attrs {
+ if strings.EqualFold(attr.Name.Local, key) {
+ return strings.TrimSpace(attr.Value), true
+ }
+ }
+ return "", false
+}
+
+func attrOrEmpty(attrs []xml.Attr, key string) string {
+ value, _ := attrValue(attrs, key)
+ return value
+}
+
+func (p *ttmlParser) collectElementText(start xml.StartElement) (string, error) {
+ var text strings.Builder
+
+ for {
+ token, err := p.decoder.Token()
+ if err != nil {
+ return "", err
+ }
+
+ switch t := token.(type) {
+ case xml.StartElement:
+ value, err := p.collectElementText(t)
+ if err != nil {
+ return "", err
+ }
+ text.WriteString(value)
+ case xml.EndElement:
+ if strings.EqualFold(t.Name.Local, start.Name.Local) {
+ return text.String(), nil
+ }
+ case xml.CharData:
+ text.WriteString(string(t))
+ }
+ }
+}
+
+func (p *ttmlParser) skipElement(_ xml.StartElement) error {
+ depth := 1
+ for depth > 0 {
+ token, err := p.decoder.Token()
+ if err != nil {
+ return err
+ }
+
+ switch token.(type) {
+ case xml.StartElement:
+ depth++
+ case xml.EndElement:
+ depth--
+ }
+ }
+ return nil
+}
+
+func normalizeTTMLLang(lang string) string {
+ lang = strings.ToLower(strings.TrimSpace(lang))
+ if lang == "" {
+ return "xxx"
+ }
+ return lang
+}
+
+func sanitizeTTMLText(raw string) string {
+ raw = str.SanitizeText(raw)
+ raw = strings.ReplaceAll(raw, "\r\n", "\n")
+ raw = strings.ReplaceAll(raw, "\r", "\n")
+
+ lines := strings.Split(raw, "\n")
+ for i := range lines {
+ lines[i] = strings.TrimSpace(lines[i])
+ }
+ return strings.TrimSpace(strings.Join(lines, "\n"))
+}
+
+func linesAreSynced(lines []model.Line) bool {
+ for i := range lines {
+ if lines[i].Start != nil {
+ return true
+ }
+ for j := range lines[i].Cue {
+ if lines[i].Cue[j].Start != nil {
+ return true
+ }
+ }
+ }
+ return false
+}
+
+func hydrateLineTimingFromTokens(line model.Line) model.Line {
+ return model.NormalizeLineTiming(line)
+}
+
+func positiveOrDefault(v float64, fallback float64) float64 {
+ if v <= 0 {
+ return fallback
+ }
+ return v
+}
diff --git a/core/lyrics/ttml_test.go b/core/lyrics/ttml_test.go
new file mode 100644
index 000000000..07f41a080
--- /dev/null
+++ b/core/lyrics/ttml_test.go
@@ -0,0 +1,430 @@
+package lyrics
+
+import (
+ "github.com/navidrome/navidrome/model"
+ . "github.com/onsi/ginkgo/v2"
+ . "github.com/onsi/gomega"
+)
+
+var _ = Describe("parseTTML", func() {
+ Describe("Multi-language and timing", func() {
+ It("should parse multiple language divs with inherited offsets and frame/tick timing", func() {
+ content := []byte(`
+
+
+
+
Line one
+
Line two
with break
+
+
+
+`)
+
+ list, err := parseTTML(content)
+ Expect(err).ToNot(HaveOccurred())
+ Expect(list).To(HaveLen(2))
+
+ By("parsing the English track")
+ eng := list[0]
+ Expect(eng.Lang).To(Equal("eng"))
+ Expect(eng.Synced).To(BeTrue())
+ Expect(eng.Line[0].Start).To(Equal(ptr(int64(3000))))
+ Expect(eng.Line[0].Value).To(Equal("Line one"))
+ Expect(eng.Line[1].Start).To(Equal(ptr(int64(4517))))
+ Expect(eng.Line[1].Value).To(Equal("Line two\nwith break"))
+
+ By("parsing the Portuguese track")
+ por := list[1]
+ Expect(por.Lang).To(Equal("por"))
+ Expect(por.Line[0].Start).To(Equal(ptr(int64(4500))))
+ Expect(por.Line[0].Value).To(Equal("Linha"))
+ })
+ })
+
+ Describe("Unsupported cue handling", func() {
+ It("should skip wallclock cues and keep valid ones", func() {
+ content := []byte(`
+
+
+
+
+`)
+
+ list, err := parseTTML(content)
+ Expect(err).ToNot(HaveOccurred())
+ Expect(list).To(HaveLen(1))
+ Expect(list[0].Line).To(HaveLen(1))
+ Expect(list[0].Line[0].Start).To(Equal(ptr(int64(1000))))
+ Expect(list[0].Line[0].Value).To(Equal("Keep me"))
+ })
+ })
+
+ Describe("Begin/End/Dur with inheritance", func() {
+ It("should correctly accumulate nested timing from body, div, and p elements", func() {
+ content := []byte(`
+
+
+
+
First line
+
Second line
+
+
+`)
+
+ list, err := parseTTML(content)
+ Expect(err).ToNot(HaveOccurred())
+ Expect(list).To(HaveLen(1))
+ Expect(list[0].Lang).To(Equal("eng"))
+ Expect(list[0].Line).To(HaveLen(2))
+ Expect(list[0].Line[0].Start).To(Equal(ptr(int64(16000))))
+ Expect(list[0].Line[0].Value).To(Equal("First line"))
+ Expect(list[0].Line[1].Start).To(Equal(ptr(int64(18000))))
+ Expect(list[0].Line[1].Value).To(Equal("Second line"))
+ })
+ })
+
+ Describe("Non-standard bare second offsets", func() {
+ It("should parse bare decimal numbers as seconds", func() {
+ content := []byte(`
+
+
+
+
First line
+
Second line
+
+
+`)
+
+ list, err := parseTTML(content)
+ Expect(err).ToNot(HaveOccurred())
+ Expect(list).To(HaveLen(1))
+ Expect(list[0].Line).To(HaveLen(2))
+ Expect(list[0].Line[0].Start).To(Equal(ptr(int64(10170))))
+ Expect(list[0].Line[0].Value).To(Equal("First line"))
+ Expect(list[0].Line[1].Start).To(Equal(ptr(int64(13710))))
+ Expect(list[0].Line[1].Value).To(Equal("Second line"))
+ })
+ })
+
+ Describe("Word timing tokens", func() {
+ It("should extract timed tokens from spans including background role", func() {
+ content := []byte(`
+
+
+
+
+`)
+
+ list, err := parseTTML(content)
+ Expect(err).ToNot(HaveOccurred())
+ Expect(list).To(HaveLen(1))
+ Expect(list[0].Agents).To(Equal([]model.Agent{
+ {ID: "main", Role: "main"},
+ {ID: "__nd_bg__|main", Role: "bg"},
+ }))
+ Expect(list[0].Line).To(HaveLen(1))
+
+ line := list[0].Line[0]
+ Expect(line.Start).To(Equal(ptr(int64(1000))))
+ Expect(line.Value).To(Equal("Hello\necho"))
+ Expect(line.End).To(Equal(ptr(int64(3000))))
+ Expect(line.Cue).To(HaveLen(3))
+
+ Expect(line.Cue[0]).To(Equal(model.Cue{Start: ptr(int64(1000)), End: ptr(int64(1400)), Value: "He", ByteStart: 0, ByteEnd: 1, AgentID: "main"}))
+ Expect(line.Cue[1]).To(Equal(model.Cue{Start: ptr(int64(1400)), End: ptr(int64(1800)), Value: "llo", ByteStart: 2, ByteEnd: 4, AgentID: "main"}))
+ Expect(line.Cue[2]).To(Equal(model.Cue{Start: ptr(int64(2000)), End: ptr(int64(2500)), Value: "echo", ByteStart: 6, ByteEnd: 9, AgentID: "__nd_bg__|main"}))
+ })
+
+ It("should append role tokens exactly instead of using substring matches", func() {
+ content := []byte(`
+
+
+
+
+`)
+
+ list, err := parseTTML(content)
+
+ Expect(err).ToNot(HaveOccurred())
+ Expect(list).To(HaveLen(1))
+ Expect(list[0].Agents).To(Equal([]model.Agent{
+ {ID: "main", Role: "main"},
+ {ID: "__nd_bg__|main", Role: "bg"},
+ }))
+ Expect(list[0].Line).To(HaveLen(1))
+ Expect(list[0].Line[0].Cue).To(HaveLen(2))
+ Expect(list[0].Line[0].Cue[0].AgentID).To(Equal("main"))
+ Expect(list[0].Line[0].Cue[1].AgentID).To(Equal("__nd_bg__|main"))
+ })
+
+ It("should parse named TTML agents into main, voice, and group roles", func() {
+ content := []byte(`
+
+
+
+ Chris Martin
+ Jin
+ All
+
+
+
+
+
+`)
+
+ list, err := parseTTML(content)
+ Expect(err).ToNot(HaveOccurred())
+ Expect(list).To(HaveLen(1))
+ Expect(list[0].Agents).To(Equal([]model.Agent{
+ {ID: "v1", Role: "main", Name: "Chris Martin"},
+ {ID: "v2", Role: "voice", Name: "Jin"},
+ {ID: "v1000", Role: "group", Name: "All"},
+ }))
+ Expect(list[0].Line[0].Cue[0].AgentID).To(Equal("v1"))
+ Expect(list[0].Line[1].Cue[0].AgentID).To(Equal("v2"))
+ Expect(list[0].Line[2].Cue[0].AgentID).To(Equal("v1000"))
+ })
+
+ It("should avoid collisions between derived background agents and explicit TTML agent ids", func() {
+ content := []byte(`
+
+
+
+ Lead
+ Existing Background Id
+
+
+
+
+
+ Lead
+ Echo
+
+
+ Named
+
+
+
+`)
+
+ list, err := parseTTML(content)
+ Expect(err).ToNot(HaveOccurred())
+ Expect(list).To(HaveLen(1))
+ Expect(list[0].Agents).To(Equal([]model.Agent{
+ {ID: "lead", Role: "main", Name: "Lead"},
+ {ID: "__nd_bg__|lead", Role: "bg", Name: "Lead"},
+ {ID: "lead__bg", Role: "voice", Name: "Existing Background Id"},
+ }))
+ Expect(list[0].Line).To(HaveLen(2))
+ Expect(list[0].Line[0].Cue).To(HaveLen(2))
+ Expect(list[0].Line[0].Cue[0].AgentID).To(Equal("lead"))
+ Expect(list[0].Line[0].Cue[1].AgentID).To(Equal("__nd_bg__|lead"))
+ Expect(list[0].Line[1].Cue).To(HaveLen(1))
+ Expect(list[0].Line[1].Cue[0].AgentID).To(Equal("lead__bg"))
+ })
+
+ It("should fill missing cue agent ids with the resolved main agent", func() {
+ content := []byte(`
+
+
+
+ Guest Vocal
+
+
+
+
+
+`)
+
+ list, err := parseTTML(content)
+ Expect(err).ToNot(HaveOccurred())
+ Expect(list).To(HaveLen(1))
+ Expect(list[0].Agents).To(Equal([]model.Agent{
+ {ID: "guest", Role: "main", Name: "Guest Vocal"},
+ }))
+ Expect(list[0].Line).To(HaveLen(1))
+ Expect(list[0].Line[0].Cue).To(HaveLen(2))
+ Expect(list[0].Line[0].Cue[0].AgentID).To(Equal("guest"))
+ Expect(list[0].Line[0].Cue[1].AgentID).To(Equal("guest"))
+ })
+ })
+
+ Describe("Ambiguous decimal timing", func() {
+ It("should prefer absolute timing when values fall inside parent window", func() {
+ content := []byte(`
+
+
+
+
+`)
+
+ list, err := parseTTML(content)
+ Expect(err).ToNot(HaveOccurred())
+ Expect(list).To(HaveLen(1))
+ Expect(list[0].Line).To(HaveLen(1))
+
+ line := list[0].Line[0]
+ Expect(line.Start).To(Equal(ptr(int64(43444))))
+ Expect(line.Value).To(Equal("go\ngo"))
+ Expect(line.End).To(Equal(ptr(int64(45570))))
+ Expect(line.Cue).To(HaveLen(2))
+ Expect(line.Cue[0]).To(Equal(model.Cue{Start: ptr(int64(43444)), End: ptr(int64(43716)), Value: "go", ByteStart: 0, ByteEnd: 1}))
+ Expect(line.Cue[1]).To(Equal(model.Cue{Start: ptr(int64(43716)), End: ptr(int64(43887)), Value: "go", ByteStart: 3, ByteEnd: 4}))
+ })
+ })
+
+ Describe("Unsynced fallback", func() {
+ It("should return unsynced lyrics when no timing is present", func() {
+ content := []byte(`
+
+
+
+
+`)
+
+ list, err := parseTTML(content)
+ Expect(err).ToNot(HaveOccurred())
+ Expect(list).To(HaveLen(1))
+ Expect(list[0].Lang).To(Equal("xxx"))
+ Expect(list[0].Synced).To(BeFalse())
+ Expect(list[0].Line).To(HaveLen(1))
+ Expect(list[0].Line[0].Start).To(BeNil())
+ Expect(list[0].Line[0].Value).To(Equal("No timing here"))
+ })
+ })
+
+ Describe("Metadata tracks", func() {
+ It("should produce main, translation, and pronunciation tracks from iTunesMetadata", func() {
+ content := []byte(`
+
+
+
+
+
+
+ Hola
+ Skip me
+
+
+
+
+ konni
+
+
+
+
+
+
+
+
+`)
+
+ list, err := parseTTML(content)
+ Expect(err).ToNot(HaveOccurred())
+ Expect(list).To(HaveLen(3))
+
+ By("checking the main track")
+ main := list[0]
+ Expect(main.Kind).To(Equal("main"))
+ Expect(main.Lang).To(Equal("ja"))
+ Expect(main.Line).To(HaveLen(2))
+
+ By("checking the translation track")
+ translation := list[1]
+ Expect(translation.Kind).To(Equal("translation"))
+ Expect(translation.Lang).To(Equal("es"))
+ Expect(translation.Line).To(HaveLen(1))
+ Expect(translation.Line[0].Start).To(Equal(ptr(int64(1000))))
+ Expect(translation.Line[0].Value).To(Equal("Hola"))
+ Expect(translation.Line[0].End).To(Equal(ptr(int64(1500))))
+
+ By("checking the pronunciation track")
+ pronunciation := list[2]
+ Expect(pronunciation.Kind).To(Equal("pronunciation"))
+ Expect(pronunciation.Lang).To(Equal("ja-latn"))
+ Expect(pronunciation.Line).To(HaveLen(1))
+ Expect(pronunciation.Line[0].Start).To(Equal(ptr(int64(2000))))
+ Expect(pronunciation.Line[0].Value).To(Equal("konni"))
+ Expect(pronunciation.Line[0].End).To(Equal(ptr(int64(2600))))
+ Expect(pronunciation.Line[0].Cue).To(HaveLen(2))
+ Expect(pronunciation.Line[0].Cue[0]).To(Equal(model.Cue{Start: ptr(int64(2000)), End: ptr(int64(2300)), Value: "ko", ByteStart: 0, ByteEnd: 1}))
+ Expect(pronunciation.Line[0].Cue[1]).To(Equal(model.Cue{Start: ptr(int64(2300)), End: ptr(int64(2600)), Value: "nni", ByteStart: 2, ByteEnd: 4}))
+ })
+ })
+
+ Describe("Pronunciation with bare decimal end times", func() {
+ It("should correctly parse bare decimal times in transliteration spans", func() {
+ content := []byte(`
+
+
+
+
+
+
+ I woke up
+
+
+
+
+
+
+
+
+`)
+
+ list, err := parseTTML(content)
+ Expect(err).ToNot(HaveOccurred())
+
+ var pronunciation *model.Lyrics
+ for i := range list {
+ if list[i].Kind == "pronunciation" {
+ pronunciation = &list[i]
+ break
+ }
+ }
+ Expect(pronunciation).ToNot(BeNil())
+ Expect(pronunciation.Line).To(HaveLen(1))
+
+ line := pronunciation.Line[0]
+ Expect(line.Start).To(Equal(ptr(int64(2747))))
+ Expect(line.Value).To(Equal("I woke up"))
+ Expect(line.Cue).To(HaveLen(3))
+ Expect(line.Cue[0]).To(Equal(model.Cue{Start: ptr(int64(2747)), End: ptr(int64(3018)), Value: "I", ByteStart: 0, ByteEnd: 0}))
+ Expect(line.Cue[1]).To(Equal(model.Cue{Start: ptr(int64(3018)), End: ptr(int64(3179)), Value: "woke", ByteStart: 2, ByteEnd: 5}))
+ Expect(line.Cue[2]).To(Equal(model.Cue{Start: ptr(int64(3179)), End: ptr(int64(3582)), Value: "up", ByteStart: 7, ByteEnd: 8}))
+ })
+ })
+})
diff --git a/model/lyrics.go b/model/lyrics.go
index f75f3b11b..f9f21b873 100644
--- a/model/lyrics.go
+++ b/model/lyrics.go
@@ -6,33 +6,57 @@ import (
"slices"
"strconv"
"strings"
+ "unicode"
"github.com/navidrome/navidrome/log"
"github.com/navidrome/navidrome/utils/str"
)
+type Cue struct {
+ Start *int64 `structs:"start,omitempty" json:"start,omitempty"`
+ End *int64 `structs:"end,omitempty" json:"end,omitempty"`
+ Value string `structs:"value" json:"value"`
+ ByteStart int `structs:"byteStart" json:"byteStart"`
+ ByteEnd int `structs:"byteEnd" json:"byteEnd"`
+ AgentID string `structs:"agentId,omitempty" json:"agentId,omitempty"`
+}
+
+type Agent struct {
+ ID string `structs:"id" json:"id"`
+ Role string `structs:"role" json:"role"`
+ Name string `structs:"name,omitempty" json:"name,omitempty"`
+}
+
type Line struct {
Start *int64 `structs:"start,omitempty" json:"start,omitempty"`
+ End *int64 `structs:"end,omitempty" json:"end,omitempty"`
Value string `structs:"value" json:"value"`
+ Cue []Cue `structs:"cue,omitempty" json:"cue,omitempty"`
}
type Lyrics struct {
- DisplayArtist string `structs:"displayArtist,omitempty" json:"displayArtist,omitempty"`
- DisplayTitle string `structs:"displayTitle,omitempty" json:"displayTitle,omitempty"`
- Lang string `structs:"lang" json:"lang"`
- Line []Line `structs:"line" json:"line"`
- Offset *int64 `structs:"offset,omitempty" json:"offset,omitempty"`
- Synced bool `structs:"synced" json:"synced"`
+ DisplayArtist string `structs:"displayArtist,omitempty" json:"displayArtist,omitempty"`
+ DisplayTitle string `structs:"displayTitle,omitempty" json:"displayTitle,omitempty"`
+ Kind string `structs:"kind,omitempty" json:"kind,omitempty"`
+ Lang string `structs:"lang" json:"lang"`
+ Agents []Agent `structs:"agents,omitempty" json:"agents,omitempty"`
+ Line []Line `structs:"line" json:"line"`
+ Offset *int64 `structs:"offset,omitempty" json:"offset,omitempty"`
+ Synced bool `structs:"synced" json:"synced"`
}
// support the standard [mm:ss.mm], as well as [hh:*] and [*.mmm]
-const timeRegexString = `\[([0-9]{1,2}:)?([0-9]{1,2}):([0-9]{1,2})(.[0-9]{1,3})?\]`
+const timeRegexString = `\[([0-9]{1,2}:)?([0-9]{1,2}):([0-9]{1,2})(\.[0-9]{1,3})?\]`
var (
// Should either be at the beginning of file, or beginning of line
syncRegex = regexp.MustCompile(`(^|\n)\s*` + timeRegexString)
timeRegex = regexp.MustCompile(timeRegexString)
lrcIdRegex = regexp.MustCompile(`\[(ar|ti|offset|lang):([^]]+)]`)
+
+ // Enhanced LRC: inline word-level timing markers like <00:12.34>
+ enhancedLRCTimeString = `<([0-9]{1,2}:)?([0-9]{1,2}):([0-9]{1,2})(\.[0-9]{1,3})?>`
+ enhancedLRCRegex = regexp.MustCompile(enhancedLRCTimeString)
)
func (l Lyrics) IsEmpty() bool {
@@ -106,9 +130,11 @@ func ToLyrics(language, text string) (*Lyrics, error) {
if validLine {
for idx := range timestamps {
+ value, cues := parseEnhancedLine(priorLine)
structuredLines = append(structuredLines, Line{
Start: ×tamps[idx],
- Value: strings.TrimSpace(priorLine),
+ Value: value,
+ Cue: cues,
})
}
timestamps = nil
@@ -154,9 +180,11 @@ func ToLyrics(language, text string) (*Lyrics, error) {
if validLine {
for idx := range timestamps {
+ value, cues := parseEnhancedLine(priorLine)
structuredLines = append(structuredLines, Line{
Start: ×tamps[idx],
- Value: strings.TrimSpace(priorLine),
+ Value: value,
+ Cue: cues,
})
}
}
@@ -173,13 +201,118 @@ func ToLyrics(language, text string) (*Lyrics, error) {
DisplayArtist: artist,
DisplayTitle: title,
Lang: language,
- Line: structuredLines,
+ Line: NormalizeCueLines(structuredLines),
Offset: offset,
Synced: synced,
}
return &lyrics, nil
}
+// parseEnhancedLine extracts word-level timing cues from Enhanced LRC inline markers
+// and computes UTF-8 byte offsets against the final stripped line value.
+func parseEnhancedLine(text string) (string, []Cue) {
+ matches := enhancedLRCRegex.FindAllStringSubmatchIndex(text, -1)
+ if len(matches) == 0 {
+ return strings.TrimSpace(text), nil
+ }
+
+ type segment struct {
+ start int64
+ rawStart int
+ rawEnd int
+ }
+
+ segments := make([]segment, 0, len(matches))
+ var rawValue strings.Builder
+ for i, match := range matches {
+ timeMs, err := parseTime(
+ // Rewrite <...> as [...] so parseTime can handle it with the same logic
+ "["+text[match[0]+1:match[1]-1]+"]",
+ // Adjust match indices to point into our rewritten string (need start/end pairs for each group)
+ []int{
+ 0, match[1] - match[0],
+ adjustGroup(match, 2), adjustGroup(match, 3),
+ adjustGroup(match, 4), adjustGroup(match, 5),
+ adjustGroup(match, 6), adjustGroup(match, 7),
+ adjustGroup(match, 8), adjustGroup(match, 9),
+ },
+ )
+ if err != nil {
+ continue
+ }
+
+ // Text runs from after this marker to the start of the next marker (or end of string)
+ textStart := match[1]
+ var textEnd int
+ if i+1 < len(matches) {
+ textEnd = matches[i+1][0]
+ } else {
+ textEnd = len(text)
+ }
+
+ word := text[textStart:textEnd]
+ if word == "" {
+ continue
+ }
+
+ rawStart := rawValue.Len()
+ rawValue.WriteString(word)
+ segments = append(segments, segment{
+ start: timeMs,
+ rawStart: rawStart,
+ rawEnd: rawValue.Len(),
+ })
+ }
+
+ if len(segments) == 0 {
+ return strings.TrimSpace(stripEnhancedMarkers(text)), nil
+ }
+
+ finalRaw := rawValue.String()
+ leftTrimBytes := len(finalRaw) - len(strings.TrimLeftFunc(finalRaw, unicode.IsSpace))
+ rightTrimBytes := len(finalRaw) - len(strings.TrimRightFunc(finalRaw, unicode.IsSpace))
+ trimmedEnd := len(finalRaw) - rightTrimBytes
+ if trimmedEnd < leftTrimBytes {
+ trimmedEnd = leftTrimBytes
+ }
+
+ cues := make([]Cue, 0, len(segments))
+ for _, seg := range segments {
+ start := seg.start
+ byteStart := max(seg.rawStart, leftTrimBytes)
+ byteEnd := min(seg.rawEnd, trimmedEnd)
+ if byteStart >= byteEnd {
+ continue
+ }
+
+ cues = append(cues, Cue{
+ Start: &start,
+ Value: finalRaw[byteStart:byteEnd],
+ ByteStart: byteStart - leftTrimBytes,
+ ByteEnd: byteEnd - leftTrimBytes - 1,
+ })
+ }
+
+ return strings.TrimSpace(finalRaw), cues
+}
+
+// adjustGroup remaps a capture group index from the original match to our rewritten "[...]" string.
+// The rewrite shifts by -1 (removed '<', added '[') so positions within the brackets stay the same.
+func adjustGroup(match []int, groupIdx int) int {
+ orig := match[groupIdx]
+ if orig == -1 {
+ return -1
+ }
+ // Offset is: original position minus the position of '<' in the original, plus 1 for '['
+ return orig - match[0]
+}
+
+// stripEnhancedMarkers removes all inline markers from text,
+// returning the plain lyric text.
+func stripEnhancedMarkers(text string) string {
+ return enhancedLRCRegex.ReplaceAllString(text, "")
+}
+
func parseTime(line string, match []int) (int64, error) {
var hours, millis int64
var err error
@@ -227,3 +360,119 @@ func parseTime(line string, match []int) (int64, error) {
}
type LyricList []Lyrics
+
+func NormalizeLyrics(lyrics Lyrics) Lyrics {
+ lyrics.Line = NormalizeCueLines(lyrics.Line)
+ if len(lyrics.Agents) == 0 {
+ lyrics.Agents = nil
+ }
+ return lyrics
+}
+
+func NormalizeCueLines(lines []Line) []Line {
+ if len(lines) == 0 {
+ return lines
+ }
+
+ normalized := make([]Line, len(lines))
+ copy(normalized, lines)
+
+ for i := range normalized {
+ if len(normalized[i].Cue) > 0 {
+ normalized[i].Cue = slices.Clone(normalized[i].Cue)
+ }
+
+ var fallbackEnd *int64
+ if normalized[i].End != nil {
+ v := *normalized[i].End
+ fallbackEnd = &v
+ } else if i+1 < len(normalized) && normalized[i+1].Start != nil {
+ v := *normalized[i+1].Start
+ fallbackEnd = &v
+ }
+
+ normalized[i] = normalizeCueLine(normalized[i], fallbackEnd)
+ }
+
+ return normalized
+}
+
+func NormalizeLineTiming(line Line) Line {
+ if len(line.Cue) == 0 {
+ return line
+ }
+
+ var earliestStart *int64
+ var latestEnd *int64
+ for i := range line.Cue {
+ token := line.Cue[i]
+ if token.Start != nil {
+ if earliestStart == nil || *token.Start < *earliestStart {
+ v := *token.Start
+ earliestStart = &v
+ }
+ }
+
+ candidateEnd := token.End
+ if candidateEnd == nil {
+ candidateEnd = token.Start
+ }
+ if candidateEnd != nil {
+ if latestEnd == nil || *candidateEnd > *latestEnd {
+ v := *candidateEnd
+ latestEnd = &v
+ }
+ }
+ }
+
+ if line.Start == nil && earliestStart != nil {
+ v := *earliestStart
+ line.Start = &v
+ }
+ if line.End == nil && latestEnd != nil {
+ v := *latestEnd
+ line.End = &v
+ }
+ return line
+}
+
+func normalizeCueLine(line Line, fallbackEnd *int64) Line {
+ if len(line.Cue) == 0 {
+ return line
+ }
+
+ for i := range line.Cue {
+ if line.Cue[i].End != nil {
+ continue
+ }
+
+ if i+1 < len(line.Cue) && line.Cue[i+1].Start != nil {
+ v := *line.Cue[i+1].Start
+ line.Cue[i].End = &v
+ continue
+ }
+
+ if fallbackEnd != nil {
+ v := *fallbackEnd
+ line.Cue[i].End = &v
+ }
+ }
+
+ for i := range line.Cue {
+ if line.Cue[i].End == nil {
+ line.Cue = clearCueEnds(line.Cue)
+ return NormalizeLineTiming(line)
+ }
+ }
+
+ return NormalizeLineTiming(line)
+}
+
+func clearCueEnds(cues []Cue) []Cue {
+ normalized := make([]Cue, len(cues))
+ copy(normalized, cues)
+ for i := range normalized {
+ normalized[i].End = nil
+ }
+ return normalized
+}
diff --git a/model/lyrics_test.go b/model/lyrics_test.go
index 644b85ad2..21bdf0e3d 100644
--- a/model/lyrics_test.go
+++ b/model/lyrics_test.go
@@ -108,4 +108,121 @@ var _ = Describe("ToLyrics", func() {
{Start: new(int64(1000 * 60 * 60 * 51)), Value: "Test"},
}))
})
+
+ It("should parse Enhanced LRC with word-level timing", func() {
+ lyrics, err := ToLyrics("xxx", "[00:01.00]<00:01.00>Some <00:01.50>lyrics <00:02.00>here\n[00:03.00]<00:03.00>More <00:03.50>words")
+ Expect(err).ToNot(HaveOccurred())
+ Expect(lyrics.Synced).To(BeTrue())
+ Expect(lyrics.Line).To(HaveLen(2))
+
+ t1000, t1500, t2000, t3000, t3500 := int64(1000), int64(1500), int64(2000), int64(3000), int64(3500)
+
+ line0 := lyrics.Line[0]
+ Expect(line0.Start).To(Equal(&t1000))
+ Expect(line0.End).To(Equal(&t3000))
+ Expect(line0.Value).To(Equal("Some lyrics here"))
+ Expect(line0.Cue).To(Equal([]Cue{
+ {Start: &t1000, End: &t1500, Value: "Some ", ByteStart: 0, ByteEnd: 4},
+ {Start: &t1500, End: &t2000, Value: "lyrics ", ByteStart: 5, ByteEnd: 11},
+ {Start: &t2000, End: &t3000, Value: "here", ByteStart: 12, ByteEnd: 15},
+ }))
+
+ line1 := lyrics.Line[1]
+ Expect(line1.Start).To(Equal(&t3000))
+ Expect(line1.End).To(Equal(&t3500))
+ Expect(line1.Value).To(Equal("More words"))
+ Expect(line1.Cue).To(Equal([]Cue{
+ {Start: &t3000, Value: "More ", ByteStart: 0, ByteEnd: 4},
+ {Start: &t3500, Value: "words", ByteStart: 5, ByteEnd: 9},
+ }))
+
+ Expect(line1.Cue[1].End).To(BeNil())
+ })
+
+ It("should not parse malformed Enhanced LRC timing markers", func() {
+ lyrics, err := ToLyrics("xxx", "[00:01.00]<00:01a50>Not a marker")
+ Expect(err).ToNot(HaveOccurred())
+ Expect(lyrics.Synced).To(BeTrue())
+ Expect(lyrics.Line).To(Equal([]Line{
+ {Start: new(int64(1000)), Value: "<00:01a50>Not a marker"},
+ }))
+ })
+
+ It("should ignore Enhanced LRC markers and return plain lines when no markers present", func() {
+ a, b := int64(1000), int64(3000)
+ lyrics, err := ToLyrics("xxx", "[00:01.00]Plain line\n[00:03.00]Another plain line")
+ Expect(err).ToNot(HaveOccurred())
+ Expect(lyrics.Line).To(Equal([]Line{
+ {Start: &a, Value: "Plain line"},
+ {Start: &b, Value: "Another plain line"},
+ }))
+ })
+
+ It("should handle mixed Enhanced and plain LRC lines", func() {
+ lyrics, err := ToLyrics("xxx", "[00:01.00]<00:01.00>Some <00:01.50>lyrics\n[00:03.00]Plain line\n[00:05.00]<00:05.00>More <00:05.50>words")
+ Expect(err).ToNot(HaveOccurred())
+ Expect(lyrics.Line).To(HaveLen(3))
+
+ t1000, t1500, t5000, t5500 := int64(1000), int64(1500), int64(5000), int64(5500)
+ t3000 := int64(3000)
+
+ Expect(lyrics.Line[0].Cue).To(Equal([]Cue{
+ {Start: &t1000, End: &t1500, Value: "Some ", ByteStart: 0, ByteEnd: 4},
+ {Start: &t1500, End: &t3000, Value: "lyrics", ByteStart: 5, ByteEnd: 10},
+ }))
+ Expect(lyrics.Line[0].Value).To(Equal("Some lyrics"))
+ Expect(lyrics.Line[0].End).To(Equal(&t3000))
+
+ Expect(lyrics.Line[1].Cue).To(BeNil())
+ Expect(lyrics.Line[1].Value).To(Equal("Plain line"))
+
+ Expect(lyrics.Line[2].Cue).To(Equal([]Cue{
+ {Start: &t5000, Value: "More ", ByteStart: 0, ByteEnd: 4},
+ {Start: &t5500, Value: "words", ByteStart: 5, ByteEnd: 9},
+ }))
+ Expect(lyrics.Line[2].Value).To(Equal("More words"))
+ })
+
+ It("should preserve byte offsets for Enhanced LRC cues", func() {
+ lyrics, err := ToLyrics("xxx", "[00:00.00]<00:00.00>Oh <00:00.90>love<00:01.30> me <00:01.60>tonight")
+ Expect(err).ToNot(HaveOccurred())
+ Expect(lyrics.Line).To(HaveLen(1))
+
+ t0, t900, t1300, t1600 := int64(0), int64(900), int64(1300), int64(1600)
+ line := lyrics.Line[0]
+ Expect(line.Value).To(Equal("Oh love me tonight"))
+ Expect(line.Cue).To(Equal([]Cue{
+ {Start: &t0, Value: "Oh ", ByteStart: 0, ByteEnd: 2},
+ {Start: &t900, Value: "love", ByteStart: 3, ByteEnd: 6},
+ {Start: &t1300, Value: " me ", ByteStart: 7, ByteEnd: 10},
+ {Start: &t1600, Value: "tonight", ByteStart: 11, ByteEnd: 17},
+ }))
+ })
+})
+
+var _ = Describe("NormalizeCueLines", func() {
+ It("should not mutate caller cue slices when filling missing cue end times", func() {
+ start0, start1, nextLineStart := int64(1000), int64(1500), int64(3000)
+ lines := []Line{
+ {
+ Start: &start0,
+ Value: "Some lyrics",
+ Cue: []Cue{
+ {Start: &start0, Value: "Some ", ByteStart: 0, ByteEnd: 4},
+ {Start: &start1, Value: "lyrics", ByteStart: 5, ByteEnd: 10},
+ },
+ },
+ {
+ Start: &nextLineStart,
+ Value: "Next line",
+ },
+ }
+
+ normalized := NormalizeCueLines(lines)
+
+ Expect(normalized[0].Cue[0].End).To(Equal(&start1))
+ Expect(normalized[0].Cue[1].End).To(Equal(&nextLineStart))
+ Expect(lines[0].Cue[0].End).To(BeNil())
+ Expect(lines[0].Cue[1].End).To(BeNil())
+ })
})
diff --git a/model/metadata/map_mediafile.go b/model/metadata/map_mediafile.go
index 824cad7c2..ecb0a9195 100644
--- a/model/metadata/map_mediafile.go
+++ b/model/metadata/map_mediafile.go
@@ -8,6 +8,7 @@ import (
"strconv"
"github.com/navidrome/navidrome/conf"
+ lyricssvc "github.com/navidrome/navidrome/core/lyrics"
"github.com/navidrome/navidrome/log"
"github.com/navidrome/navidrome/model"
"github.com/navidrome/navidrome/utils/str"
@@ -129,7 +130,7 @@ func (md Metadata) mapGain(rg, r128 model.TagName) *float64 {
}
func (md Metadata) mapLyrics() string {
- rawLyrics := md.Pairs(model.TagLyrics)
+ rawLyrics := md.rawPairs(model.TagLyrics)
lyricList := make(model.LyricList, 0, len(rawLyrics))
@@ -137,13 +138,15 @@ func (md Metadata) mapLyrics() string {
lang := raw.Key()
text := raw.Value()
- lyrics, err := model.ToLyrics(lang, text)
+ lyrics, err := lyricssvc.ParseEmbedded(lang, text)
if err != nil {
log.Warn("Unexpected failure occurred when parsing lyrics", "file", md.filePath, err)
continue
}
- if !lyrics.IsEmpty() {
- lyricList = append(lyricList, *lyrics)
+ for _, lyric := range lyrics {
+ if !lyric.IsEmpty() {
+ lyricList = append(lyricList, lyric)
+ }
}
}
diff --git a/model/metadata/map_mediafile_test.go b/model/metadata/map_mediafile_test.go
index 16142f526..486386718 100644
--- a/model/metadata/map_mediafile_test.go
+++ b/model/metadata/map_mediafile_test.go
@@ -4,6 +4,7 @@ import (
"encoding/json"
"os"
"sort"
+ "strings"
"github.com/navidrome/navidrome/model"
"github.com/navidrome/navidrome/model/metadata"
@@ -116,5 +117,86 @@ var _ = Describe("ToMediaFile", func() {
sort.Slice(expected, func(i, j int) bool { return expected[i].Lang < expected[j].Lang })
Expect(actual).To(Equal(expected))
})
+
+ It("should parse embedded TTML lyrics before sanitizing XML tags", func() {
+ mf = toMediaFile(model.RawTags{
+ "LYRICS:ENG": {`
+
+
+
+`},
+ })
+ var actual model.LyricList
+ err := json.Unmarshal([]byte(mf.Lyrics), &actual)
+ Expect(err).ToNot(HaveOccurred())
+
+ Expect(actual).To(Equal(model.LyricList{
+ {
+ Kind: "main",
+ Lang: "eng",
+ Line: []model.Line{{Start: ptr(int64(1000)), End: ptr(int64(2500)), Value: "Embedded TTML line"}},
+ Synced: true,
+ },
+ }))
+ })
+
+ It("should parse embedded TTML lyrics longer than the metadata tag max length", func() {
+ padding := strings.Repeat(`padding`, 1400)
+ content := `
+
+
+
+
+ ` + padding + `
+
+
+
+
+
+
+
Long embedded TTML line
+
+
+`
+
+ Expect(len(content)).To(BeNumerically(">", 32768))
+
+ mf = toMediaFile(model.RawTags{
+ "LYRICS:ENG": {content},
+ })
+ var actual model.LyricList
+ err := json.Unmarshal([]byte(mf.Lyrics), &actual)
+ Expect(err).ToNot(HaveOccurred())
+
+ Expect(actual).To(HaveLen(1))
+ Expect(actual[0].Kind).To(Equal("main"))
+ Expect(actual[0].Lang).To(Equal("en"))
+ Expect(actual[0].Line).To(Equal([]model.Line{
+ {Start: ptr(int64(1000)), End: ptr(int64(2500)), Value: "Long embedded TTML line"},
+ }))
+ })
+
+ It("should parse embedded SRT lyrics with the tag language", func() {
+ mf = toMediaFile(model.RawTags{
+ "LYRICS:POR": {`1
+00:00:18,800 --> 00:00:22,800
+Estamos nas legendas`},
+ })
+ var actual model.LyricList
+ err := json.Unmarshal([]byte(mf.Lyrics), &actual)
+ Expect(err).ToNot(HaveOccurred())
+
+ Expect(actual).To(Equal(model.LyricList{
+ {
+ Lang: "por",
+ Line: []model.Line{
+ {Start: ptr(int64(18800)), End: ptr(int64(22800)), Value: "Estamos nas legendas"},
+ },
+ Synced: true,
+ },
+ }))
+ })
})
})
diff --git a/model/metadata/metadata.go b/model/metadata/metadata.go
index 48928f989..c62f33776 100644
--- a/model/metadata/metadata.go
+++ b/model/metadata/metadata.go
@@ -70,6 +70,7 @@ func New(filePath string, info Info) Metadata {
return Metadata{
filePath: filePath,
fileInfo: info.FileInfo,
+ rawTags: lowerTags(info.Tags),
tags: clean(filePath, info.Tags),
audioProps: info.AudioProperties,
hasPicture: info.HasPicture,
@@ -79,6 +80,7 @@ func New(filePath string, info Info) Metadata {
type Metadata struct {
filePath string
fileInfo FileInfo
+ rawTags model.Tags
tags model.Tags
audioProps AudioProperties
hasPicture bool
@@ -114,6 +116,14 @@ func (md Metadata) Pairs(key model.TagName) []Pair {
values := md.tags[key]
return slice.Map(values, func(v string) Pair { return Pair(v) })
}
+func (md Metadata) rawPairs(key model.TagName) []Pair {
+ mapping, ok := model.TagMappings()[key]
+ if !ok {
+ return nil
+ }
+ values := filterDuplicatedOrEmptyValues(processPairMapping(key, mapping, md.rawTags))
+ return slice.Map(values, func(v string) Pair { return Pair(v) })
+}
func (md Metadata) first(key model.TagName) string {
if v, ok := md.tags[key]; ok && len(v) > 0 {
return v[0]
diff --git a/model/metadata/test_helpers_test.go b/model/metadata/test_helpers_test.go
new file mode 100644
index 000000000..abb65dfdb
--- /dev/null
+++ b/model/metadata/test_helpers_test.go
@@ -0,0 +1,5 @@
+package metadata_test
+
+func ptr[T any](v T) *T {
+ return &v
+}
diff --git a/server/subsonic/helpers.go b/server/subsonic/helpers.go
index e4c39e373..565b064a2 100644
--- a/server/subsonic/helpers.go
+++ b/server/subsonic/helpers.go
@@ -494,14 +494,79 @@ func mapExplicitStatus(explicitStatus string) string {
return ""
}
-func buildStructuredLyric(mf *model.MediaFile, lyrics model.Lyrics) responses.StructuredLyric {
+func buildStructuredLyric(mf *model.MediaFile, lyrics model.Lyrics, enhanced bool) responses.StructuredLyric {
lines := make([]responses.Line, len(lyrics.Line))
+ var cueLines []responses.CueLine
+ agentOrderByID := make(map[string]int, len(lyrics.Agents))
+ agentRoleByID := make(map[string]string, len(lyrics.Agents))
+ responseAgents := make([]responses.Agent, 0, len(lyrics.Agents))
+
+ for i, agent := range lyrics.Agents {
+ agentOrderByID[agent.ID] = i
+ agentRoleByID[agent.ID] = agent.Role
+ responseAgents = append(responseAgents, responses.Agent{
+ ID: agent.ID,
+ Role: agent.Role,
+ Name: agent.Name,
+ })
+ }
for i, line := range lyrics.Line {
lines[i] = responses.Line{
Start: line.Start,
Value: line.Value,
}
+ if !enhanced || len(line.Cue) == 0 {
+ continue
+ }
+
+ agentOrder := make([]string, 0, 2)
+ cuesByAgent := make(map[string][]model.Cue)
+ for _, cue := range line.Cue {
+ if cue.Start == nil {
+ continue
+ }
+ agentID := strings.TrimSpace(cue.AgentID)
+ if _, exists := cuesByAgent[agentID]; !exists {
+ agentOrder = append(agentOrder, agentID)
+ }
+ cuesByAgent[agentID] = append(cuesByAgent[agentID], cue)
+ }
+
+ sort.SliceStable(agentOrder, func(i, j int) bool {
+ leftRole := agentRoleByID[agentOrder[i]]
+ rightRole := agentRoleByID[agentOrder[j]]
+ if leftRole == "main" && rightRole != "main" {
+ return true
+ }
+ if rightRole == "main" && leftRole != "main" {
+ return false
+ }
+
+ leftOrder, leftOK := agentOrderByID[agentOrder[i]]
+ rightOrder, rightOK := agentOrderByID[agentOrder[j]]
+ if leftOK && rightOK && leftOrder != rightOrder {
+ return leftOrder < rightOrder
+ }
+ if leftOK != rightOK {
+ return leftOK
+ }
+ return i < j
+ })
+
+ for _, agentID := range agentOrder {
+ cueLine := responses.CueLine{
+ Index: int32(i),
+ Start: line.Start,
+ End: line.End,
+ Value: line.Value,
+ Cue: buildLyricCues(cuesByAgent[agentID], line.End),
+ }
+ if agentID != "" {
+ cueLine.AgentID = agentID
+ }
+ cueLines = append(cueLines, cueLine)
+ }
}
structured := responses.StructuredLyric{
@@ -509,10 +574,22 @@ func buildStructuredLyric(mf *model.MediaFile, lyrics model.Lyrics) responses.St
DisplayTitle: lyrics.DisplayTitle,
Lang: lyrics.Lang,
Line: lines,
+ CueLine: cueLines,
Offset: lyrics.Offset,
Synced: lyrics.Synced,
}
+ if enhanced {
+ kind := strings.TrimSpace(lyrics.Kind)
+ if kind == "" {
+ kind = "main"
+ }
+ structured.Kind = kind
+ if len(cueLines) > 0 && len(responseAgents) > 0 {
+ structured.Agents = responseAgents
+ }
+ }
+
if structured.DisplayArtist == "" {
structured.DisplayArtist = mf.Artist
}
@@ -523,11 +600,86 @@ func buildStructuredLyric(mf *model.MediaFile, lyrics model.Lyrics) responses.St
return structured
}
-func buildLyricsList(mf *model.MediaFile, lyricsList model.LyricList) *responses.LyricsList {
- lyricList := make(responses.StructuredLyrics, len(lyricsList))
+func buildLyricCues(cues []model.Cue, lineEnd *int64) []responses.LyricCue {
+ if len(cues) == 0 {
+ return nil
+ }
- for i, lyrics := range lyricsList {
- lyricList[i] = buildStructuredLyric(mf, lyrics)
+ hasAnyEnd := false
+ for i := range cues {
+ if cues[i].End != nil {
+ hasAnyEnd = true
+ break
+ }
+ }
+
+ normalized := make([]responses.LyricCue, 0, len(cues))
+ for i := range cues {
+ if cues[i].Start == nil {
+ continue
+ }
+
+ cue := responses.LyricCue{
+ Start: *cues[i].Start,
+ Value: cues[i].Value,
+ ByteStart: cues[i].ByteStart,
+ ByteEnd: cues[i].ByteEnd,
+ }
+ if hasAnyEnd {
+ end := cues[i].End
+ if end == nil {
+ if i+1 < len(cues) && cues[i+1].Start != nil {
+ v := *cues[i+1].Start
+ end = &v
+ } else if lineEnd != nil {
+ v := *lineEnd
+ end = &v
+ }
+ }
+ if end != nil && i+1 < len(cues) && cues[i+1].Start != nil && *end > *cues[i+1].Start {
+ v := *cues[i+1].Start
+ end = &v
+ }
+ if end != nil && *end < cue.Start {
+ v := cue.Start
+ end = &v
+ }
+ cue.End = end
+ }
+ normalized = append(normalized, cue)
+ }
+
+ if hasAnyEnd {
+ for i := range normalized {
+ if normalized[i].End == nil {
+ for j := range normalized {
+ normalized[j].End = nil
+ }
+ break
+ }
+ }
+ }
+
+ return normalized
+}
+
+func buildLyricsList(mf *model.MediaFile, lyricsList model.LyricList, enhanced bool) *responses.LyricsList {
+ var filtered model.LyricList
+ if enhanced {
+ filtered = lyricsList
+ } else {
+ // Without enhanced, only return "main" kind entries
+ for _, l := range lyricsList {
+ kind := strings.TrimSpace(l.Kind)
+ if kind == "" || kind == "main" {
+ filtered = append(filtered, l)
+ }
+ }
+ }
+
+ lyricList := make(responses.StructuredLyrics, len(filtered))
+ for i, lyrics := range filtered {
+ lyricList[i] = buildStructuredLyric(mf, lyrics, enhanced)
}
res := &responses.LyricsList{
diff --git a/server/subsonic/media_retrieval.go b/server/subsonic/media_retrieval.go
index 9ab3a20b0..8461723ed 100644
--- a/server/subsonic/media_retrieval.go
+++ b/server/subsonic/media_retrieval.go
@@ -10,6 +10,7 @@ import (
"github.com/navidrome/navidrome/conf"
"github.com/navidrome/navidrome/consts"
+ lyricssvc "github.com/navidrome/navidrome/core/lyrics"
"github.com/navidrome/navidrome/log"
"github.com/navidrome/navidrome/model"
"github.com/navidrome/navidrome/resources"
@@ -19,6 +20,8 @@ import (
"github.com/navidrome/navidrome/utils/req"
)
+const maxLegacyLyricsCandidates = 10
+
func (api *Router) GetAvatar(w http.ResponseWriter, r *http.Request) (*responses.Subsonic, error) {
if !conf.Server.EnableGravatar {
return api.getPlaceHolderAvatar(w, r)
@@ -98,7 +101,11 @@ func (api *Router) GetLyrics(r *http.Request) (*responses.Subsonic, error) {
response := newResponse()
lyricsResponse := responses.Lyrics{}
response.Lyrics = &lyricsResponse
- mediaFiles, err := api.ds.MediaFile(r.Context()).GetAll(filter.SongsByArtistTitleWithLyricsFirst(artist, title))
+ opts := filter.SongsByArtistTitleWithLyricsFirst(artist, title)
+ // Search a bounded duplicate window so source-priority fallback can still
+ // reach older matches without turning legacy getLyrics into an unbounded scan.
+ opts.Max = maxLegacyLyricsCandidates
+ mediaFiles, err := api.ds.MediaFile(r.Context()).GetAll(opts)
if err != nil {
return nil, err
@@ -108,9 +115,22 @@ func (api *Router) GetLyrics(r *http.Request) (*responses.Subsonic, error) {
return response, nil
}
- structuredLyrics, err := api.lyrics.GetLyrics(r.Context(), &mediaFiles[0])
- if err != nil {
- return nil, err
+ var structuredLyrics model.LyricList
+ if batchLyrics, ok := api.lyrics.(lyricssvc.BatchLyrics); ok {
+ structuredLyrics, err = batchLyrics.GetLyricsForMediaFiles(r.Context(), mediaFiles)
+ if err != nil {
+ return nil, err
+ }
+ } else {
+ for i := range mediaFiles {
+ structuredLyrics, err = api.lyrics.GetLyrics(r.Context(), &mediaFiles[i])
+ if err != nil {
+ return nil, err
+ }
+ if len(structuredLyrics) > 0 {
+ break
+ }
+ }
}
if len(structuredLyrics) == 0 {
@@ -124,7 +144,6 @@ func (api *Router) GetLyrics(r *http.Request) (*responses.Subsonic, error) {
for _, line := range structuredLyrics[0].Line {
lyricsText.WriteString(line.Value + "\n")
}
-
lyricsResponse.Value = lyricsText.String()
return response, nil
@@ -146,8 +165,10 @@ func (api *Router) GetLyricsBySongId(r *http.Request) (*responses.Subsonic, erro
return nil, err
}
+ enhanced, _ := req.Params(r).Bool("enhanced")
+
response := newResponse()
- response.LyricsList = buildLyricsList(mediaFile, structuredLyrics)
+ response.LyricsList = buildLyricsList(mediaFile, structuredLyrics, enhanced)
return response, nil
}
diff --git a/server/subsonic/media_retrieval_test.go b/server/subsonic/media_retrieval_test.go
index 12c0dff56..68c9cf1e6 100644
--- a/server/subsonic/media_retrieval_test.go
+++ b/server/subsonic/media_retrieval_test.go
@@ -180,6 +180,41 @@ var _ = Describe("MediaRetrievalController", func() {
Expect(response.Lyrics.Title).To(Equal("Never Gonna Give You Up"))
Expect(response.Lyrics.Value).To(Equal("We're no strangers to love\nYou know the rules and so do I\n"))
})
+
+ It("should prefer higher-priority sidecar lyrics across duplicate candidates", func() {
+ conf.Server.LyricsPriority = ".ttml,embedded"
+ r := newGetRequest("artist=Rick+Astley", "title=Never+Gonna+Give+You+Up")
+ baseTime := time.Date(2025, 1, 1, 0, 0, 0, 0, time.UTC)
+ embedded, err := model.ToLyrics("eng", "Newest duplicate embedded lyrics")
+ Expect(err).ToNot(HaveOccurred())
+ embeddedJSON, err := json.Marshal(model.LyricList{*embedded})
+ Expect(err).ToNot(HaveOccurred())
+ mockRepo.SetData(model.MediaFiles{
+ {
+ ID: "1",
+ Path: "tests/fixtures/01 Invisible (RED) Edit Version.mp3",
+ Artist: "Rick Astley",
+ Title: "Never Gonna Give You Up",
+ Lyrics: string(embeddedJSON),
+ UpdatedAt: baseTime.Add(2 * time.Hour), // Newer duplicate with embedded lyrics only
+ },
+ {
+ ID: "2",
+ Path: "tests/fixtures/test.mp3",
+ Artist: "Rick Astley",
+ Title: "Never Gonna Give You Up",
+ Lyrics: "[]",
+ UpdatedAt: baseTime.Add(1 * time.Hour), // Older, but has TTML sidecar
+ },
+ })
+
+ response, err := router.GetLyrics(r)
+ Expect(err).ToNot(HaveOccurred())
+ Expect(response.Lyrics.Artist).To(Equal("Rick Astley"))
+ Expect(response.Lyrics.Title).To(Equal("Never Gonna Give You Up"))
+ Expect(response.Lyrics.Value).To(Equal("We're no strangers to love\nYou know the rules and so do I\n"))
+ Expect(mockRepo.Options.Max).To(Equal(maxLegacyLyricsCandidates))
+ })
})
Describe("GetLyricsBySongId", func() {
@@ -196,8 +231,10 @@ var _ = Describe("MediaRetrievalController", func() {
Expect(realLyric.DisplayArtist).To(Equal(expectedLyric.DisplayArtist))
Expect(realLyric.DisplayTitle).To(Equal(expectedLyric.DisplayTitle))
+ Expect(realLyric.Kind).To(Equal(expectedLyric.Kind))
Expect(realLyric.Lang).To(Equal(expectedLyric.Lang))
Expect(realLyric.Synced).To(Equal(expectedLyric.Synced))
+ Expect(realLyric.Agents).To(Equal(expectedLyric.Agents))
if expectedLyric.Offset == nil {
Expect(realLyric.Offset).To(BeNil())
@@ -216,6 +253,38 @@ var _ = Describe("MediaRetrievalController", func() {
Expect(*realLine.Start).To(Equal(*expectedLine.Start))
}
}
+
+ Expect(realLyric.CueLine).To(HaveLen(len(expectedLyric.CueLine)))
+ for j, realCueLine := range realLyric.CueLine {
+ expectedCueLine := expectedLyric.CueLine[j]
+ Expect(realCueLine.Index).To(Equal(expectedCueLine.Index))
+ Expect(realCueLine.Value).To(Equal(expectedCueLine.Value))
+ Expect(realCueLine.AgentID).To(Equal(expectedCueLine.AgentID))
+ if expectedCueLine.Start == nil {
+ Expect(realCueLine.Start).To(BeNil())
+ } else {
+ Expect(*realCueLine.Start).To(Equal(*expectedCueLine.Start))
+ }
+ if expectedCueLine.End == nil {
+ Expect(realCueLine.End).To(BeNil())
+ } else {
+ Expect(*realCueLine.End).To(Equal(*expectedCueLine.End))
+ }
+
+ Expect(realCueLine.Cue).To(HaveLen(len(expectedCueLine.Cue)))
+ for k, realCue := range realCueLine.Cue {
+ expectedCue := expectedCueLine.Cue[k]
+ Expect(realCue.Value).To(Equal(expectedCue.Value))
+ Expect(realCue.Start).To(Equal(expectedCue.Start))
+ Expect(realCue.ByteStart).To(Equal(expectedCue.ByteStart))
+ Expect(realCue.ByteEnd).To(Equal(expectedCue.ByteEnd))
+ if expectedCue.End == nil {
+ Expect(realCue.End).To(BeNil())
+ } else {
+ Expect(*realCue.End).To(Equal(*expectedCue.End))
+ }
+ }
+ }
}
}
@@ -316,6 +385,427 @@ var _ = Describe("MediaRetrievalController", func() {
},
})
})
+
+ It("should return multilingual TTML sidecar lyrics", func() {
+ conf.Server.LyricsPriority = ".ttml,embedded"
+ r := newGetRequest("id=1")
+
+ mockRepo.SetData(model.MediaFiles{
+ {
+ ID: "1",
+ Path: "tests/fixtures/test.mp3",
+ Artist: "Rick Astley",
+ Title: "Never Gonna Give You Up",
+ Lyrics: "[]",
+ },
+ })
+
+ response, err := router.GetLyricsBySongId(r)
+ Expect(err).ToNot(HaveOccurred())
+
+ porTime := int64(18800)
+ ttmlTime := int64(22800)
+ compareResponses(response.LyricsList, responses.LyricsList{
+ StructuredLyrics: responses.StructuredLyrics{
+ {
+ DisplayArtist: "Rick Astley",
+ DisplayTitle: "Never Gonna Give You Up",
+ Lang: "eng",
+ Synced: true,
+ Line: []responses.Line{
+ {
+ Start: ×[0],
+ Value: "We're no strangers to love",
+ },
+ {
+ Start: &ttmlTime,
+ Value: "You know the rules and so do I",
+ },
+ },
+ },
+ {
+ DisplayArtist: "Rick Astley",
+ DisplayTitle: "Never Gonna Give You Up",
+ Lang: "por",
+ Synced: true,
+ Line: []responses.Line{
+ {
+ Start: &porTime,
+ Value: "Nao somos estranhos ao amor",
+ },
+ },
+ },
+ },
+ })
+ })
+
+ It("should return metadata-linked translation and pronunciation tracks from TTML", func() {
+ conf.Server.LyricsPriority = ".ttml,embedded"
+ r := newGetRequest("id=1&enhanced=true")
+
+ mockRepo.SetData(model.MediaFiles{
+ {
+ ID: "1",
+ Path: "tests/fixtures/test-metadata.mp3",
+ Artist: "Rick Astley",
+ Title: "Never Gonna Give You Up",
+ Lyrics: "[]",
+ },
+ })
+
+ response, err := router.GetLyricsBySongId(r)
+ Expect(err).ToNot(HaveOccurred())
+
+ mainStartA := int64(1000)
+ mainStartB := int64(2000)
+ tokenStartA := int64(2000)
+ tokenEndA := int64(2300)
+ tokenStartB := int64(2300)
+ tokenEndB := int64(2600)
+ compareResponses(response.LyricsList, responses.LyricsList{
+ StructuredLyrics: responses.StructuredLyrics{
+ {
+ DisplayArtist: "Rick Astley",
+ DisplayTitle: "Never Gonna Give You Up",
+ Kind: "main",
+ Lang: "ja",
+ Synced: true,
+ Line: []responses.Line{
+ {
+ Start: &mainStartA,
+ Value: "こんにちは",
+ },
+ {
+ Start: &mainStartB,
+ Value: "こんばんは",
+ },
+ },
+ },
+ {
+ DisplayArtist: "Rick Astley",
+ DisplayTitle: "Never Gonna Give You Up",
+ Kind: "translation",
+ Lang: "es",
+ Synced: true,
+ Line: []responses.Line{
+ {
+ Start: &mainStartA,
+ Value: "Hola",
+ },
+ },
+ },
+ {
+ DisplayArtist: "Rick Astley",
+ DisplayTitle: "Never Gonna Give You Up",
+ Kind: "pronunciation",
+ Lang: "ja-latn",
+ Synced: true,
+ Line: []responses.Line{
+ {
+ Start: &mainStartB,
+ Value: "konni",
+ },
+ },
+ CueLine: []responses.CueLine{
+ {
+ Index: 0,
+ Start: &mainStartB,
+ End: &tokenEndB,
+ Value: "konni",
+ Cue: []responses.LyricCue{
+ {
+ Start: tokenStartA,
+ End: &tokenEndA,
+ ByteStart: 0,
+ ByteEnd: 1,
+ Value: "ko",
+ },
+ {
+ Start: tokenStartB,
+ End: &tokenEndB,
+ ByteStart: 2,
+ ByteEnd: 4,
+ Value: "nni",
+ },
+ },
+ },
+ },
+ },
+ },
+ })
+ })
+
+ It("should return cue lines for songLyrics v2 clients with enhanced=true", func() {
+ r := newGetRequest("id=1&enhanced=true")
+
+ lineStart := int64(1000)
+ lineEnd := int64(3000)
+ tokenStartA := int64(1000)
+ tokenEndA := int64(1400)
+ tokenStartB := int64(2000)
+ tokenEndB := int64(2500)
+ lyricsJson, err := json.Marshal(model.LyricList{
+ {
+ Lang: "eng",
+ Agents: []model.Agent{{ID: "lead", Role: "main"}, {ID: "__nd_bg__|lead", Role: "bg"}},
+ Synced: true,
+ Line: []model.Line{
+ {
+ Start: &lineStart,
+ End: &lineEnd,
+ Value: "Hello echo",
+ Cue: []model.Cue{
+ {
+ Start: &tokenStartA,
+ End: &tokenEndA,
+ Value: "Hello",
+ ByteStart: 0,
+ ByteEnd: 4,
+ AgentID: "lead",
+ },
+ {
+ Start: &tokenStartB,
+ End: &tokenEndB,
+ Value: "echo",
+ ByteStart: 6,
+ ByteEnd: 9,
+ AgentID: "__nd_bg__|lead",
+ },
+ },
+ },
+ },
+ },
+ })
+ Expect(err).ToNot(HaveOccurred())
+
+ mockRepo.SetData(model.MediaFiles{
+ {
+ ID: "1",
+ Artist: "Rick Astley",
+ Title: "Never Gonna Give You Up",
+ Lyrics: string(lyricsJson),
+ },
+ })
+
+ response, err := router.GetLyricsBySongId(r)
+ Expect(err).ToNot(HaveOccurred())
+ compareResponses(response.LyricsList, responses.LyricsList{
+ StructuredLyrics: responses.StructuredLyrics{
+ {
+ DisplayArtist: "Rick Astley",
+ DisplayTitle: "Never Gonna Give You Up",
+ Kind: "main",
+ Lang: "eng",
+ Synced: true,
+ Agents: []responses.Agent{
+ {ID: "lead", Role: "main"},
+ {ID: "__nd_bg__|lead", Role: "bg"},
+ },
+ Line: []responses.Line{
+ {
+ Start: &lineStart,
+ Value: "Hello echo",
+ },
+ },
+ CueLine: []responses.CueLine{
+ {
+ Index: 0,
+ Start: &lineStart,
+ End: &lineEnd,
+ Value: "Hello echo",
+ AgentID: "lead",
+ Cue: []responses.LyricCue{
+ {
+ Start: tokenStartA,
+ End: &tokenEndA,
+ ByteStart: 0,
+ ByteEnd: 4,
+ Value: "Hello",
+ },
+ },
+ },
+ {
+ Index: 0,
+ Start: &lineStart,
+ End: &lineEnd,
+ Value: "Hello echo",
+ AgentID: "__nd_bg__|lead",
+ Cue: []responses.LyricCue{
+ {
+ Start: tokenStartB,
+ End: &tokenEndB,
+ ByteStart: 6,
+ ByteEnd: 9,
+ Value: "echo",
+ },
+ },
+ },
+ },
+ },
+ },
+ })
+ })
+
+ It("should keep enhanced line-level lyrics when no cue data is available", func() {
+ r := newGetRequest("id=1&enhanced=true")
+
+ lineStart := int64(1000)
+ lineEnd := int64(3000)
+ lyricsJSON, err := json.Marshal(model.LyricList{
+ {
+ Kind: "main",
+ Lang: "eng",
+ Synced: true,
+ Line: []model.Line{
+ {
+ Start: &lineStart,
+ End: &lineEnd,
+ Value: "Line without word timing",
+ },
+ },
+ },
+ })
+ Expect(err).ToNot(HaveOccurred())
+
+ mockRepo.SetData(model.MediaFiles{
+ {
+ ID: "1",
+ Artist: "Rick Astley",
+ Title: "Never Gonna Give You Up",
+ Lyrics: string(lyricsJSON),
+ },
+ })
+
+ response, err := router.GetLyricsBySongId(r)
+ Expect(err).ToNot(HaveOccurred())
+ compareResponses(response.LyricsList, responses.LyricsList{
+ StructuredLyrics: responses.StructuredLyrics{
+ {
+ DisplayArtist: "Rick Astley",
+ DisplayTitle: "Never Gonna Give You Up",
+ Kind: "main",
+ Lang: "eng",
+ Synced: true,
+ Line: []responses.Line{
+ {
+ Start: &lineStart,
+ Value: "Line without word timing",
+ },
+ },
+ },
+ },
+ })
+ })
+
+ It("should return required cue byte offsets for ambiguous and multibyte cue lines", func() {
+ r := newGetRequest("id=1&enhanced=true")
+
+ asciiLineStart := int64(0)
+ asciiLineEnd := int64(2400)
+ asciiCueStartA := int64(0)
+ asciiCueEndA := int64(300)
+ asciiCueStartB := int64(900)
+ asciiCueEndB := int64(1300)
+ asciiCueStartC := int64(1300)
+ asciiCueEndC := int64(1600)
+ asciiCueStartD := int64(1600)
+
+ utfLineStart := int64(2747)
+ utfLineEnd := int64(6214)
+ utfCueStartA := int64(2747)
+ utfCueEndA := int64(3018)
+ utfCueStartB := int64(3018)
+ utfCueEndB := int64(3179)
+ utfCueStartC := int64(3582)
+ utfCueEndC := int64(4100)
+ utfCueStartD := int64(4500)
+ utfCueEndD := int64(6214)
+
+ lyricsJSON, err := json.Marshal(model.LyricList{
+ {
+ Lang: "eng",
+ Synced: true,
+ Line: []model.Line{
+ {
+ Start: &asciiLineStart,
+ End: &asciiLineEnd,
+ Value: "Oh love love me tonight",
+ Cue: []model.Cue{
+ {Start: &asciiCueStartA, End: &asciiCueEndA, Value: "Oh", ByteStart: 0, ByteEnd: 1},
+ {Start: &asciiCueStartB, End: &asciiCueEndB, Value: "love", ByteStart: 8, ByteEnd: 11},
+ {Start: &asciiCueStartC, End: &asciiCueEndC, Value: "me", ByteStart: 13, ByteEnd: 14},
+ {Start: &asciiCueStartD, Value: "tonight", ByteStart: 16, ByteEnd: 22},
+ },
+ },
+ {
+ Start: &utfLineStart,
+ End: &utfLineEnd,
+ Value: "눈을 뜬 순간",
+ Cue: []model.Cue{
+ {Start: &utfCueStartA, End: &utfCueEndA, Value: "눈", ByteStart: 0, ByteEnd: 2},
+ {Start: &utfCueStartB, End: &utfCueEndB, Value: "을", ByteStart: 3, ByteEnd: 5},
+ {Start: &utfCueStartC, End: &utfCueEndC, Value: "뜬", ByteStart: 7, ByteEnd: 9},
+ {Start: &utfCueStartD, End: &utfCueEndD, Value: "순간", ByteStart: 11, ByteEnd: 16},
+ },
+ },
+ },
+ },
+ })
+ Expect(err).ToNot(HaveOccurred())
+
+ mockRepo.SetData(model.MediaFiles{
+ {
+ ID: "1",
+ Artist: "Rick Astley",
+ Title: "Never Gonna Give You Up",
+ Lyrics: string(lyricsJSON),
+ },
+ })
+
+ response, err := router.GetLyricsBySongId(r)
+ Expect(err).ToNot(HaveOccurred())
+ compareResponses(response.LyricsList, responses.LyricsList{
+ StructuredLyrics: responses.StructuredLyrics{
+ {
+ DisplayArtist: "Rick Astley",
+ DisplayTitle: "Never Gonna Give You Up",
+ Kind: "main",
+ Lang: "eng",
+ Synced: true,
+ Line: []responses.Line{
+ {Start: &asciiLineStart, Value: "Oh love love me tonight"},
+ {Start: &utfLineStart, Value: "눈을 뜬 순간"},
+ },
+ CueLine: []responses.CueLine{
+ {
+ Index: 0,
+ Start: &asciiLineStart,
+ End: &asciiLineEnd,
+ Value: "Oh love love me tonight",
+ Cue: []responses.LyricCue{
+ {Start: asciiCueStartA, End: &asciiCueEndA, Value: "Oh", ByteStart: 0, ByteEnd: 1},
+ {Start: asciiCueStartB, End: &asciiCueEndB, Value: "love", ByteStart: 8, ByteEnd: 11},
+ {Start: asciiCueStartC, End: &asciiCueEndC, Value: "me", ByteStart: 13, ByteEnd: 14},
+ {Start: asciiCueStartD, End: &asciiLineEnd, Value: "tonight", ByteStart: 16, ByteEnd: 22},
+ },
+ },
+ {
+ Index: 1,
+ Start: &utfLineStart,
+ End: &utfLineEnd,
+ Value: "눈을 뜬 순간",
+ Cue: []responses.LyricCue{
+ {Start: utfCueStartA, End: &utfCueEndA, Value: "눈", ByteStart: 0, ByteEnd: 2},
+ {Start: utfCueStartB, End: &utfCueEndB, Value: "을", ByteStart: 3, ByteEnd: 5},
+ {Start: utfCueStartC, End: &utfCueEndC, Value: "뜬", ByteStart: 7, ByteEnd: 9},
+ {Start: utfCueStartD, End: &utfCueEndD, Value: "순간", ByteStart: 11, ByteEnd: 16},
+ },
+ },
+ },
+ },
+ },
+ })
+ })
})
})
diff --git a/server/subsonic/opensubsonic.go b/server/subsonic/opensubsonic.go
index 85edb1012..97b3cafcc 100644
--- a/server/subsonic/opensubsonic.go
+++ b/server/subsonic/opensubsonic.go
@@ -11,7 +11,7 @@ func (api *Router) GetOpenSubsonicExtensions(_ *http.Request) (*responses.Subson
extensions := responses.OpenSubsonicExtensions{
{Name: "transcodeOffset", Versions: []int32{1}},
{Name: "formPost", Versions: []int32{1}},
- {Name: "songLyrics", Versions: []int32{1}},
+ {Name: "songLyrics", Versions: []int32{1, 2}},
{Name: "indexBasedQueue", Versions: []int32{1}},
{Name: "transcoding", Versions: []int32{1}},
{Name: "playbackReport", Versions: []int32{1}},
diff --git a/server/subsonic/opensubsonic_test.go b/server/subsonic/opensubsonic_test.go
index 3ccbf232e..e4217303f 100644
--- a/server/subsonic/opensubsonic_test.go
+++ b/server/subsonic/opensubsonic_test.go
@@ -58,7 +58,7 @@ var _ = Describe("GetOpenSubsonicExtensions", func() {
HaveLen(6),
ContainElement(responses.OpenSubsonicExtension{Name: "transcodeOffset", Versions: []int32{1}}),
ContainElement(responses.OpenSubsonicExtension{Name: "formPost", Versions: []int32{1}}),
- ContainElement(responses.OpenSubsonicExtension{Name: "songLyrics", Versions: []int32{1}}),
+ ContainElement(responses.OpenSubsonicExtension{Name: "songLyrics", Versions: []int32{1, 2}}),
ContainElement(responses.OpenSubsonicExtension{Name: "indexBasedQueue", Versions: []int32{1}}),
ContainElement(responses.OpenSubsonicExtension{Name: "transcoding", Versions: []int32{1}}),
ContainElement(responses.OpenSubsonicExtension{Name: "playbackReport", Versions: []int32{1}}),
@@ -88,7 +88,7 @@ var _ = Describe("GetOpenSubsonicExtensions", func() {
HaveLen(7),
ContainElement(responses.OpenSubsonicExtension{Name: "transcodeOffset", Versions: []int32{1}}),
ContainElement(responses.OpenSubsonicExtension{Name: "formPost", Versions: []int32{1}}),
- ContainElement(responses.OpenSubsonicExtension{Name: "songLyrics", Versions: []int32{1}}),
+ ContainElement(responses.OpenSubsonicExtension{Name: "songLyrics", Versions: []int32{1, 2}}),
ContainElement(responses.OpenSubsonicExtension{Name: "indexBasedQueue", Versions: []int32{1}}),
ContainElement(responses.OpenSubsonicExtension{Name: "transcoding", Versions: []int32{1}}),
ContainElement(responses.OpenSubsonicExtension{Name: "playbackReport", Versions: []int32{1}}),
diff --git a/server/subsonic/responses/responses.go b/server/subsonic/responses/responses.go
index dcb458932..7e41a1daa 100644
--- a/server/subsonic/responses/responses.go
+++ b/server/subsonic/responses/responses.go
@@ -547,13 +547,39 @@ type Line struct {
Value string `xml:",chardata" json:"value"`
}
+type LyricCue struct {
+ Start int64 `xml:"start,attr" json:"start"`
+ End *int64 `xml:"end,attr,omitempty" json:"end,omitempty"`
+ ByteStart int `xml:"byteStart,attr" json:"byteStart"`
+ ByteEnd int `xml:"byteEnd,attr" json:"byteEnd"`
+ Value string `xml:",chardata" json:"value"`
+}
+
+type Agent struct {
+ ID string `xml:"id,attr" json:"id"`
+ Role string `xml:"role,attr" json:"role"`
+ Name string `xml:"name,attr,omitempty" json:"name,omitempty"`
+}
+
+type CueLine struct {
+ Index int32 `xml:"index,attr" json:"index"`
+ Start *int64 `xml:"start,attr,omitempty" json:"start,omitempty"`
+ End *int64 `xml:"end,attr,omitempty" json:"end,omitempty"`
+ Value string `xml:"value,attr" json:"value"`
+ AgentID string `xml:"agentId,attr,omitempty" json:"agentId,omitempty"`
+ Cue []LyricCue `xml:"cue,omitempty" json:"cue,omitempty"`
+}
+
type StructuredLyric struct {
- DisplayArtist string `xml:"displayArtist,attr,omitempty" json:"displayArtist,omitempty"`
- DisplayTitle string `xml:"displayTitle,attr,omitempty" json:"displayTitle,omitempty"`
- Lang string `xml:"lang,attr" json:"lang"`
- Line []Line `xml:"line" json:"line"`
- Offset *int64 `xml:"offset,attr,omitempty" json:"offset,omitempty"`
- Synced bool `xml:"synced,attr" json:"synced"`
+ DisplayArtist string `xml:"displayArtist,attr,omitempty" json:"displayArtist,omitempty"`
+ DisplayTitle string `xml:"displayTitle,attr,omitempty" json:"displayTitle,omitempty"`
+ Kind string `xml:"kind,attr,omitempty" json:"kind,omitempty"`
+ Lang string `xml:"lang,attr" json:"lang"`
+ Line []Line `xml:"line" json:"line"`
+ Agents []Agent `xml:"agent,omitempty" json:"agents,omitempty"`
+ CueLine []CueLine `xml:"cueLine,omitempty" json:"cueLine,omitempty"`
+ Offset *int64 `xml:"offset,attr,omitempty" json:"offset,omitempty"`
+ Synced bool `xml:"synced,attr" json:"synced"`
}
type StructuredLyrics []StructuredLyric
diff --git a/tests/fixtures/bom-test.ttml b/tests/fixtures/bom-test.ttml
new file mode 100644
index 000000000..319ab1f07
--- /dev/null
+++ b/tests/fixtures/bom-test.ttml
@@ -0,0 +1,2 @@
+
+
diff --git a/tests/fixtures/bom-utf16-test.ttml b/tests/fixtures/bom-utf16-test.ttml
new file mode 100644
index 000000000..a5621ef5d
Binary files /dev/null and b/tests/fixtures/bom-utf16-test.ttml differ
diff --git a/tests/fixtures/test-enhanced.lrc b/tests/fixtures/test-enhanced.lrc
new file mode 100644
index 000000000..8f7b60f8c
--- /dev/null
+++ b/tests/fixtures/test-enhanced.lrc
@@ -0,0 +1,6 @@
+[ar:Test Artist]
+[ti:Enhanced Test]
+[lang:eng]
+[00:01.00]<00:01.00>Some <00:01.50>lyrics <00:02.00>here
+[00:03.00]<00:03.00>More <00:03.50>words
+[00:05.00]Plain line without inline markers
diff --git a/tests/fixtures/test-metadata.ttml b/tests/fixtures/test-metadata.ttml
new file mode 100644
index 000000000..c0243c18f
--- /dev/null
+++ b/tests/fixtures/test-metadata.ttml
@@ -0,0 +1,25 @@
+
+
+
+
+
+
+
+ Hola
+
+
+
+
+ konni
+
+
+
+
+
+
+
+
+
diff --git a/tests/fixtures/test.elrc b/tests/fixtures/test.elrc
new file mode 100644
index 000000000..01c3d2cdd
--- /dev/null
+++ b/tests/fixtures/test.elrc
@@ -0,0 +1,5 @@
+[ar:ELRC Artist]
+[ti:ELRC Song]
+[lang:eng]
+[00:01.00]<00:01.00>Lead <00:01.50>words
+[00:03.00]Fallback line
diff --git a/tests/fixtures/test.srt b/tests/fixtures/test.srt
new file mode 100644
index 000000000..3c9c09a39
--- /dev/null
+++ b/tests/fixtures/test.srt
@@ -0,0 +1,7 @@
+1
+00:00:18,800 --> 00:00:22,800
+We're from subtitles
+
+2
+00:00:22,801 --> 00:00:26,000
+Another subtitle line
diff --git a/tests/fixtures/test.ttml b/tests/fixtures/test.ttml
new file mode 100644
index 000000000..a85673a1b
--- /dev/null
+++ b/tests/fixtures/test.ttml
@@ -0,0 +1,12 @@
+
+
+
+
+
We're no strangers to love
+
You know the rules and so do I
+
+
+
Nao somos estranhos ao amor
+
+
+