From 34e4b94251117b8049b5e7a955e01ea2b81364ea Mon Sep 17 00:00:00 2001 From: Deluan Date: Wed, 19 Aug 2026 23:41:35 -0400 Subject: [PATCH] fix(scanner): stop splitting multi-byte characters when truncating tags sanitize() capped tag values with a byte slice, so a value whose limit falls in the middle of a multi-byte character was stored as invalid UTF-8. defaultMaxTagLength is 1024, which is not a multiple of 3, so any sufficiently long CJK title hit this. Only trailing invalid bytes are trimmed, leaving bad bytes elsewhere in the value untouched. --- model/metadata/metadata.go | 9 +++++++++ model/metadata/metadata_test.go | 22 ++++++++++++++++++++++ 2 files changed, 31 insertions(+) diff --git a/model/metadata/metadata.go b/model/metadata/metadata.go index 729e83564..8cffdea7a 100644 --- a/model/metadata/metadata.go +++ b/model/metadata/metadata.go @@ -9,6 +9,7 @@ import ( "strconv" "strings" "time" + "unicode/utf8" "github.com/google/uuid" "github.com/navidrome/navidrome/consts" @@ -366,6 +367,14 @@ func sanitize(filePath string, tagName model.TagName, tag model.TagConf, value s if len(value) > maxLength { log.Trace("Truncated tag value", "tag", tagName, "value", value, "length", len(value), "maxLength", maxLength) value = value[:maxLength] + // Drop the trailing partial rune the cut may have left behind. Only trailing + // invalid bytes are removed, so pre-existing bad bytes elsewhere are preserved. + for len(value) > 0 { + if r, size := utf8.DecodeLastRuneInString(value); r != utf8.RuneError || size > 1 { + break + } + value = value[:len(value)-1] + } } switch tag.Type { diff --git a/model/metadata/metadata_test.go b/model/metadata/metadata_test.go index 7ebe9fa4a..346fb8755 100644 --- a/model/metadata/metadata_test.go +++ b/model/metadata/metadata_test.go @@ -4,6 +4,7 @@ import ( "os" "strings" "time" + "unicode/utf8" "github.com/navidrome/navidrome/model" "github.com/navidrome/navidrome/model/metadata" @@ -122,6 +123,27 @@ var _ = Describe("Metadata", func() { Expect(pair[0].Value()).To(HaveLen(1048570)) }) + It("should not split a multi-byte character when truncating", func() { + // 1024 is not a multiple of 3, so a byte-wise cut lands mid-rune. + props.Tags = model.RawTags{ + "Title": {strings.Repeat("日", 2048)}, + } + md = metadata.New(filePath, props) + + title := md.String(model.TagTitle) + Expect(utf8.ValidString(title)).To(BeTrue(), "truncation produced invalid UTF-8") + Expect(len(title)).To(BeNumerically("<=", 1024)) + }) + + It("should keep invalid bytes that are not at the truncation point", func() { + props.Tags = model.RawTags{ + "Title": {"a\xffb" + strings.Repeat("c", 2048)}, + } + md = metadata.New(filePath, props) + + Expect(md.String(model.TagTitle)).To(HaveLen(1024)) + }) + It("should split multiple values", func() { props.Tags = model.RawTags{ "Genre": {"Rock/Pop;;Punk"},