From 6c50697a1a5a64928b6abed5c570f57a31e15444 Mon Sep 17 00:00:00 2001 From: ShukeBta <272197458+ShukeBta@users.noreply.github.com> Date: Thu, 18 Jun 2026 02:58:30 +0800 Subject: [PATCH] Fix cloud media metadata repair and merging --- cmd/server/main.go | 6 ++ internal/repository/repository.go | 25 ++++++ internal/service/cloud_metadata.go | 10 +++ internal/service/cloud_path_repair.go | 109 ++++++++++++++++++++++++++ internal/service/external_id_hints.go | 96 +++++++++++++++++++++++ internal/service/local_metadata.go | 6 ++ internal/service/media.go | 10 +++ internal/service/media_test.go | 107 +++++++++++++++++++++++++ internal/service/scanner.go | 52 +++++++++++- internal/service/scraper.go | 43 ++++++++++ internal/service/scraper_test.go | 63 +++++++++++++++ web/src/utils/groupSeries.ts | 5 ++ 12 files changed, 530 insertions(+), 2 deletions(-) create mode 100644 internal/service/cloud_path_repair.go create mode 100644 internal/service/external_id_hints.go diff --git a/cmd/server/main.go b/cmd/server/main.go index 80594ef..80e4e08 100644 --- a/cmd/server/main.go +++ b/cmd/server/main.go @@ -86,6 +86,12 @@ func main() { applyCPUThreadLimit(cfg, logger) services := service.New(cfg, logger, repos) + if repaired, err := services.RepairCloudPathMetadata(context.Background()); err != nil { + logger.Warn("cloud path metadata repair failed", zap.Error(err)) + } else if repaired > 0 { + logger.Info("cloud path metadata repair completed", zap.Int("media_count", repaired)) + } + if err := services.Auth.SeedAdmin(context.Background()); err != nil { logger.Warn("seed admin failed", zap.Error(err)) } diff --git a/internal/repository/repository.go b/internal/repository/repository.go index c37e5be..9d98c92 100644 --- a/internal/repository/repository.go +++ b/internal/repository/repository.go @@ -377,6 +377,31 @@ func (r *MediaRepository) upsert(ctx context.Context, m *model.Media) error { } } } + if existing.ScrapeStatus == "pending" || existing.ScrapeStatus == "" || existing.ScrapeStatus == "no_match" { + backfilledExternalID := false + if m.TMDbID > 0 && existing.TMDbID <= 0 { + updates["tm_db_id"] = m.TMDbID + backfilledExternalID = true + } + if m.BangumiID > 0 && existing.BangumiID <= 0 { + updates["bangumi_id"] = m.BangumiID + backfilledExternalID = true + } + if m.DoubanID != "" && existing.DoubanID == "" { + updates["douban_id"] = m.DoubanID + backfilledExternalID = true + } + if m.TheTVDBID != "" && existing.TheTVDBID == "" { + updates["thetvdb_id"] = m.TheTVDBID + backfilledExternalID = true + } + if m.Year > 0 && existing.Year <= 0 { + updates["year"] = m.Year + } + if backfilledExternalID && existing.ScrapeStatus == "no_match" { + updates["scrape_status"] = "pending" + } + } if m.ScrapeStatus == "matched" { setIfChanged(updates, "scrape_status", existing.ScrapeStatus, m.ScrapeStatus) if m.OriginalName != "" { diff --git a/internal/service/cloud_metadata.go b/internal/service/cloud_metadata.go index 13c7271..dd97dfe 100644 --- a/internal/service/cloud_metadata.go +++ b/internal/service/cloud_metadata.go @@ -49,6 +49,9 @@ func newCloudSidecarSet(typ string, entries []cloud.FileEntry) cloudSidecarSet { func (s *ScannerService) cloudDirectoryMetadata(ctx context.Context, typ, displayDir string, sidecars cloudSidecarSet, inherited *LocalMetadata) *LocalMetadata { meta := cloneLocalMetadata(inherited) + if hinted, _ := pathHintMetadata(displayDir, true); hinted != nil { + meta = mergeCloudMetadata(meta, hinted) + } for _, name := range cloudShowNFOCandidates(displayDir) { ref := sidecars.nfoByName[strings.ToLower(name)] if ref == "" { @@ -73,6 +76,9 @@ func (s *ScannerService) cloudFileMetadata(ctx context.Context, typ, displayPath season, episode := ParseEpisode(displayPath) seriesLike = seriesLike || season > 0 || episode > 0 meta := cloneLocalMetadata(inherited) + if hinted, _ := pathHintMetadata(displayPath, seriesLike); hinted != nil { + meta = mergeCloudMetadata(meta, hinted) + } base := strings.ToLower(strings.TrimSpace(strings.TrimSuffix(fileName, filepath.Ext(fileName)))) if ref := sidecars.nfoByBase[base]; ref != "" { if local, doc, err := s.readCloudNFO(ctx, typ, ref, seriesLike); err == nil && local != nil { @@ -206,6 +212,9 @@ func mergeCloudMetadata(dst, src *LocalMetadata) *LocalMetadata { if src.TMDbID > 0 { dst.TMDbID = src.TMDbID } + if src.BangumiID > 0 { + dst.BangumiID = src.BangumiID + } if src.DoubanID != "" { dst.DoubanID = src.DoubanID } @@ -230,6 +239,7 @@ func mergeCloudMetadata(dst, src *LocalMetadata) *LocalMetadata { dst.NSFW = dst.NSFW || src.NSFW dst.HasNFO = dst.HasNFO || src.HasNFO dst.HasArtwork = dst.HasArtwork || src.HasArtwork + dst.PathHint = dst.PathHint || src.PathHint return dst } diff --git a/internal/service/cloud_path_repair.go b/internal/service/cloud_path_repair.go new file mode 100644 index 0000000..dc1bf04 --- /dev/null +++ b/internal/service/cloud_path_repair.go @@ -0,0 +1,109 @@ +package service + +import ( + "context" + "strings" + + "go.uber.org/zap" + "gorm.io/gorm" + + "github.com/ShukeBta/MediaStationGo/internal/model" +) + +// RepairCloudPathMetadata backfills external IDs from cloud paths such as +// "Movie (2025) {tmdb-123}" so existing placeholder rows can be scraped +// without requiring another successful cloud provider traversal. +func (c *Container) RepairCloudPathMetadata(ctx context.Context) (int, error) { + if c == nil || c.Repo == nil || c.Repo.DB == nil { + return 0, nil + } + var repaired int + var rows []model.Media + query := c.Repo.DB.WithContext(ctx). + Model(&model.Media{}). + Select("id, title, path, year, season_num, episode_num, scrape_status, tm_db_id, bangumi_id, douban_id, thetvdb_id"). + Where("path LIKE ?", "cloud://%"). + Where("("+strings.Join([]string{ + "LOWER(path) LIKE ?", + "LOWER(path) LIKE ?", + "LOWER(path) LIKE ?", + "LOWER(path) LIKE ?", + "LOWER(path) LIKE ?", + "LOWER(path) LIKE ?", + "LOWER(path) LIKE ?", + "LOWER(path) LIKE ?", + }, " OR ")+")", + "%tmdb%", "%tmdbid%", "%douban%", "%db%", "%bangumi%", "%bgm%", "%thetvdb%", "%tvdb%") + + err := query.FindInBatches(&rows, 500, func(_ *gorm.DB, _ int) error { + for _, row := range rows { + meta, hints := pathHintMetadata(row.Path, row.SeasonNum > 0 || row.EpisodeNum > 0) + if meta == nil || !hints.useful() { + continue + } + updates := map[string]any{} + status := strings.TrimSpace(row.ScrapeStatus) + enrichable := status == "" || status == "pending" || status == "no_match" + backfilledExternalID := false + if meta.TMDbID > 0 && row.TMDbID <= 0 { + updates["tm_db_id"] = meta.TMDbID + backfilledExternalID = true + } + if meta.BangumiID > 0 && row.BangumiID <= 0 { + updates["bangumi_id"] = meta.BangumiID + backfilledExternalID = true + } + if strings.TrimSpace(meta.DoubanID) != "" && strings.TrimSpace(row.DoubanID) == "" { + updates["douban_id"] = strings.TrimSpace(meta.DoubanID) + backfilledExternalID = true + } + if strings.TrimSpace(meta.TheTVDBID) != "" && strings.TrimSpace(row.TheTVDBID) == "" { + updates["thetvdb_id"] = strings.TrimSpace(meta.TheTVDBID) + backfilledExternalID = true + } + if meta.Year > 0 && row.Year <= 0 { + updates["year"] = meta.Year + } + if enrichable && strings.TrimSpace(meta.Title) != "" && cloudPathRepairShouldReplaceTitle(row.Title, meta.Title) { + updates["title"] = strings.TrimSpace(meta.Title) + } + if backfilledExternalID && status == "no_match" { + updates["scrape_status"] = "pending" + } + if len(updates) == 0 { + continue + } + if err := c.Repo.DB.WithContext(ctx).Model(&model.Media{}).Where("id = ?", row.ID).Updates(updates).Error; err != nil { + return err + } + repaired++ + } + return nil + }).Error + if err != nil { + return repaired, err + } + if repaired > 0 && c.Log != nil { + c.Log.Info("cloud path metadata repaired", zap.Int("media_count", repaired)) + } + return repaired, nil +} + +func cloudPathRepairShouldReplaceTitle(current, hinted string) bool { + current = strings.TrimSpace(current) + hinted = strings.TrimSpace(hinted) + if hinted == "" || strings.EqualFold(current, hinted) { + return false + } + if current == "" { + return true + } + noise := []string{"web-dl", "bluray", "hdtv", "2160p", "1080p", "720p", "ddp", "aac", "h.264", "h.265", "x264", "x265", "adweb", "mweb", "cmctv", "bit"} + lower := strings.ToLower(current) + for _, token := range noise { + if strings.Contains(lower, token) { + return true + } + } + return len([]rune(current)) > len([]rune(hinted))*2 +} diff --git a/internal/service/external_id_hints.go b/internal/service/external_id_hints.go new file mode 100644 index 0000000..0250aa9 --- /dev/null +++ b/internal/service/external_id_hints.go @@ -0,0 +1,96 @@ +package service + +import ( + "regexp" + "strings" +) + +type mediaExternalIDHints struct { + TMDbID int + BangumiID int + DoubanID string + TheTVDBID string +} + +var externalIDHintPatterns = []struct { + name string + re *regexp.Regexp +}{ + {"tmdb", regexp.MustCompile(`(?i)(?:^|[^a-z0-9])(?:tmdb|tmdbid)[\s_:=#-]*(\d{2,})`)}, + {"bangumi", regexp.MustCompile(`(?i)(?:^|[^a-z0-9])(?:bangumi|bgm)[\s_:=#-]*(\d{2,})`)}, + {"douban", regexp.MustCompile(`(?i)(?:^|[^a-z0-9])(?:douban|db)[\s_:=#-]*(\d{2,})`)}, + {"thetvdb", regexp.MustCompile(`(?i)(?:^|[^a-z0-9])(?:thetvdb|tvdb)[\s_:=#-]*(\d{2,})`)}, +} + +func externalIDHintsFromText(raw string) mediaExternalIDHints { + var hints mediaExternalIDHints + for _, item := range externalIDHintPatterns { + m := item.re.FindStringSubmatch(raw) + if len(m) < 2 { + continue + } + value := strings.TrimSpace(m[1]) + switch item.name { + case "tmdb": + hints.TMDbID = mustAtoi(value) + case "bangumi": + hints.BangumiID = mustAtoi(value) + case "douban": + hints.DoubanID = value + case "thetvdb": + hints.TheTVDBID = value + } + } + return hints +} + +func (h mediaExternalIDHints) useful() bool { + return h.TMDbID > 0 || h.BangumiID > 0 || strings.TrimSpace(h.DoubanID) != "" || strings.TrimSpace(h.TheTVDBID) != "" +} + +func (h mediaExternalIDHints) applyToLocalMetadata(meta *LocalMetadata) *LocalMetadata { + if !h.useful() { + return meta + } + if meta == nil { + meta = &LocalMetadata{} + } + if h.TMDbID > 0 && meta.TMDbID <= 0 { + meta.TMDbID = h.TMDbID + } + if h.BangumiID > 0 && meta.BangumiID <= 0 { + meta.BangumiID = h.BangumiID + } + if strings.TrimSpace(h.DoubanID) != "" && meta.DoubanID == "" { + meta.DoubanID = strings.TrimSpace(h.DoubanID) + } + if strings.TrimSpace(h.TheTVDBID) != "" && meta.TheTVDBID == "" { + meta.TheTVDBID = strings.TrimSpace(h.TheTVDBID) + } + meta.PathHint = true + return meta +} + +func pathHintMetadata(raw string, _ bool) (*LocalMetadata, mediaExternalIDHints) { + hints := externalIDHintsFromText(raw) + title, year := cloudSeriesTitleFromMediaPath(raw) + if title == "" { + title, year = CleanQuery(raw) + } + if !hints.useful() && title == "" && year <= 0 { + return nil, hints + } + meta := (&mediaExternalIDHints{ + TMDbID: hints.TMDbID, + BangumiID: hints.BangumiID, + DoubanID: hints.DoubanID, + TheTVDBID: hints.TheTVDBID, + }).applyToLocalMetadata(&LocalMetadata{PathHint: true}) + if title != "" { + meta.Title = title + } + if year > 0 { + meta.Year = year + } + return meta, hints +} diff --git a/internal/service/local_metadata.go b/internal/service/local_metadata.go index 35e797a..91aa91a 100644 --- a/internal/service/local_metadata.go +++ b/internal/service/local_metadata.go @@ -25,6 +25,7 @@ type LocalMetadata struct { PosterURL string BackdropURL string TMDbID int + BangumiID int DoubanID string TheTVDBID string SeasonNum int @@ -35,6 +36,7 @@ type LocalMetadata struct { NSFW bool HasNFO bool HasArtwork bool + PathHint bool } type nfoUniqueID struct { @@ -299,6 +301,7 @@ func metadataFromDoc(doc *nfoDocument, baseDir string, seriesLike bool) *LocalMe PosterURL: firstRemoteURL(baseDir, nfoPosterValues(doc)...), BackdropURL: firstRemoteURL(baseDir, nfoBackdropValues(doc)...), TMDbID: int(doc.TMDbID), + BangumiID: mustAtoi(externalIDFromUniqueIDs(doc.UniqueIDs, "bangumi", "bgm")), DoubanID: externalIDFromUniqueIDs(doc.UniqueIDs, "douban"), TheTVDBID: externalIDFromUniqueIDs(doc.UniqueIDs, "thetvdb", "tvdb"), SeasonNum: int(doc.Season), @@ -435,6 +438,9 @@ func mergeEpisodeMetadata(dst, episode *LocalMetadata, doc *nfoDocument) { if episode.TMDbID > 0 { dst.TMDbID = episode.TMDbID } + if episode.BangumiID > 0 { + dst.BangumiID = episode.BangumiID + } if episode.DoubanID != "" { dst.DoubanID = episode.DoubanID } diff --git a/internal/service/media.go b/internal/service/media.go index 12d3875..314df28 100644 --- a/internal/service/media.go +++ b/internal/service/media.go @@ -484,6 +484,16 @@ func groupMediaVersions(items []model.Media) []MediaItem { func mediaVersionGroupKey(m model.Media) string { if m.SeasonNum > 0 || m.EpisodeNum > 0 { + switch { + case m.TMDbID > 0: + return fmt.Sprintf("episode:tmdb:%d:%d:%d", m.TMDbID, m.SeasonNum, m.EpisodeNum) + case m.BangumiID > 0: + return fmt.Sprintf("episode:bangumi:%d:%d:%d", m.BangumiID, m.SeasonNum, m.EpisodeNum) + case strings.TrimSpace(m.DoubanID) != "": + return fmt.Sprintf("episode:douban:%s:%d:%d", strings.ToLower(strings.TrimSpace(m.DoubanID)), m.SeasonNum, m.EpisodeNum) + case strings.TrimSpace(m.TheTVDBID) != "": + return fmt.Sprintf("episode:thetvdb:%s:%d:%d", strings.ToLower(strings.TrimSpace(m.TheTVDBID)), m.SeasonNum, m.EpisodeNum) + } title := firstNonEmpty(m.OriginalName, m.Title) if title == "" { title, _ = CleanQuery(m.Path) diff --git a/internal/service/media_test.go b/internal/service/media_test.go index b0ff217..a66ae45 100644 --- a/internal/service/media_test.go +++ b/internal/service/media_test.go @@ -228,6 +228,113 @@ func TestDeleteCloudLibraryPurgesMountWithoutRecycleBin(t *testing.T) { } } +func TestGroupMediaVersionsMergesEpisodeByExternalIDAcrossLibraries(t *testing.T) { + local := model.Media{ + LibraryID: "local-tv", + Title: "折腰", + Path: "/media/国产剧/折腰 (2025)/Season 1/折腰.S01E01.mkv", + SeasonNum: 1, + EpisodeNum: 1, + TMDbID: 220269, + SizeBytes: 100, + PosterURL: "https://image.tmdb.org/t/p/w500/poster.jpg", + } + cloud := model.Media{ + LibraryID: "cloud-tv", + Title: "折腰", + Path: "cloud://openlist/国产剧/折腰 (2025) {tmdb-220269}/Season 1/折腰.S01E01.mkv", + SeasonNum: 1, + EpisodeNum: 1, + TMDbID: 220269, + SizeBytes: 200, + STRMURL: "/api/cloud/play/openlist?ref=/国产剧/折腰/01.mkv", + } + + grouped := groupMediaVersions([]model.Media{local, cloud}) + if len(grouped) != 1 { + t.Fatalf("grouped len = %d, want 1: %#v", len(grouped), grouped) + } + if len(grouped[0].Versions) != 2 { + t.Fatalf("versions len = %d, want 2: %#v", len(grouped[0].Versions), grouped[0].Versions) + } +} + +func TestMediaUpsertBackfillsExternalIDsForPendingCloudRows(t *testing.T) { + db, err := gorm.Open(sqlite.Open(":memory:"), &gorm.Config{}) + if err != nil { + t.Fatal(err) + } + if err := db.AutoMigrate(&model.Media{}); err != nil { + t.Fatal(err) + } + repos := repository.New(db) + path := "cloud://openlist/国漫/折腰 (2025) {tmdb-220269}/Season 1/折腰.S01E01.mkv" + if err := repos.DB.Create(&model.Media{ + LibraryID: "cloud-tv", + Title: "折腰", + Path: path, + SeasonNum: 1, + EpisodeNum: 1, + ScrapeStatus: "pending", + }).Error; err != nil { + t.Fatal(err) + } + if err := repos.Media.Upsert(t.Context(), &model.Media{ + LibraryID: "cloud-tv", + Title: "折腰", + Path: path, + SeasonNum: 1, + EpisodeNum: 1, + TMDbID: 220269, + Year: 2025, + ScrapeStatus: "pending", + }); err != nil { + t.Fatal(err) + } + var got model.Media + if err := repos.DB.First(&got, "path = ?", path).Error; err != nil { + t.Fatal(err) + } + if got.TMDbID != 220269 || got.Year != 2025 || got.ScrapeStatus != "pending" { + t.Fatalf("pending cloud row was not backfilled correctly: tmdb=%d year=%d status=%q", got.TMDbID, got.Year, got.ScrapeStatus) + } +} + +func TestRepairCloudPathMetadataBackfillsExistingPlaceholders(t *testing.T) { + db, err := gorm.Open(sqlite.Open(":memory:"), &gorm.Config{}) + if err != nil { + t.Fatal(err) + } + if err := db.AutoMigrate(&model.Media{}); err != nil { + t.Fatal(err) + } + repos := repository.New(db) + path := "cloud://openlist/动画电影/雄狮少年2 (2024) {tmdb-1154478}/雄狮少年2 (2024) - 2160p.WEB-DL.H.265.DDP 5.1-ADWeb.mp4" + if err := repos.DB.Create(&model.Media{ + LibraryID: "cloud-movie", + Title: "雄狮少年2 adweb", + Path: path, + ScrapeStatus: "no_match", + }).Error; err != nil { + t.Fatal(err) + } + container := &Container{Repo: repos, Log: zap.NewNop()} + repaired, err := container.RepairCloudPathMetadata(t.Context()) + if err != nil { + t.Fatal(err) + } + if repaired != 1 { + t.Fatalf("repaired = %d, want 1", repaired) + } + var got model.Media + if err := repos.DB.First(&got, "path = ?", path).Error; err != nil { + t.Fatal(err) + } + if got.TMDbID != 1154478 || got.Year != 2024 || got.Title != "雄狮少年2" || got.ScrapeStatus != "pending" { + t.Fatalf("placeholder was not repaired: title=%q tmdb=%d year=%d status=%q", got.Title, got.TMDbID, got.Year, got.ScrapeStatus) + } +} + func TestSoftDeleteCloudMediaPurgesRecordWithoutRecycleBin(t *testing.T) { db, err := gorm.Open(sqlite.Open(":memory:"), &gorm.Config{}) if err != nil { diff --git a/internal/service/scanner.go b/internal/service/scanner.go index 1988dd9..9832ef7 100644 --- a/internal/service/scanner.go +++ b/internal/service/scanner.go @@ -375,6 +375,11 @@ type existingCloudMedia struct { PosterURL string BackdropURL string STRMURL string + Year int + TMDbID int + BangumiID int + DoubanID string + TheTVDBID string } type existingLocalMedia struct { @@ -1456,10 +1461,15 @@ func (s *ScannerService) existingCloudMediaSnapshot(ctx context.Context, library PosterURL string BackdropURL string STRMURL string + Year int + TMDbID int + BangumiID int + DoubanID string + TheTVDBID string } if err := s.repo.DB.WithContext(ctx). Model(&model.Media{}). - Select("path, size_bytes, duration_sec, width, height, video_codec, audio_codec, container, poster_url, backdrop_url, strm_url"). + Select("path, size_bytes, duration_sec, width, height, video_codec, audio_codec, container, poster_url, backdrop_url, strm_url, year, tm_db_id, bangumi_id, douban_id, thetvdb_id"). Where("library_id = ? AND path LIKE ?", libraryID, "cloud://%"). Find(&rows).Error; err != nil { return nil, err @@ -1478,6 +1488,11 @@ func (s *ScannerService) existingCloudMediaSnapshot(ctx context.Context, library PosterURL: row.PosterURL, BackdropURL: row.BackdropURL, STRMURL: row.STRMURL, + Year: row.Year, + TMDbID: row.TMDbID, + BangumiID: row.BangumiID, + DoubanID: row.DoubanID, + TheTVDBID: row.TheTVDBID, } } } @@ -1617,6 +1632,20 @@ func (s *ScannerService) ingestCloudFile(ctx context.Context, lib *model.Library s.queueCloudArtworkPrefetch(localMeta.PosterURL) s.queueCloudArtworkPrefetch(localMeta.BackdropURL) } + if _, hints := pathHintMetadata(path, librarySupportsSeasons(lib) || parsedSeason > 0 || parsedEpisode > 0); hints.useful() { + if hints.TMDbID > 0 && m.TMDbID <= 0 { + m.TMDbID = hints.TMDbID + } + if hints.BangumiID > 0 && m.BangumiID <= 0 { + m.BangumiID = hints.BangumiID + } + if strings.TrimSpace(hints.DoubanID) != "" && strings.TrimSpace(m.DoubanID) == "" { + m.DoubanID = strings.TrimSpace(hints.DoubanID) + } + if strings.TrimSpace(hints.TheTVDBID) != "" && strings.TrimSpace(m.TheTVDBID) == "" { + m.TheTVDBID = strings.TrimSpace(hints.TheTVDBID) + } + } if isNewMedia && writeBatch != nil { var after func() if needsTrackProbe && ext != ".strm" { @@ -1808,6 +1837,21 @@ func cloudMetadataNeedsRefresh(existing existingCloudMedia, localMeta *LocalMeta if localMeta == nil { return false } + if localMeta.Year > 0 && existing.Year <= 0 { + return true + } + if localMeta.TMDbID > 0 && existing.TMDbID <= 0 { + return true + } + if localMeta.BangumiID > 0 && existing.BangumiID <= 0 { + return true + } + if strings.TrimSpace(localMeta.DoubanID) != "" && strings.TrimSpace(existing.DoubanID) == "" { + return true + } + if strings.TrimSpace(localMeta.TheTVDBID) != "" && strings.TrimSpace(existing.TheTVDBID) == "" { + return true + } if strings.TrimSpace(localMeta.PosterURL) != "" && strings.TrimSpace(existing.PosterURL) == "" { return true } @@ -2334,6 +2378,9 @@ func applyLocalMetadata(m *model.Media, local *LocalMetadata) { if local.TMDbID > 0 { m.TMDbID = local.TMDbID } + if local.BangumiID > 0 { + m.BangumiID = local.BangumiID + } if local.DoubanID != "" { m.DoubanID = local.DoubanID } @@ -2358,7 +2405,7 @@ func applyLocalMetadata(m *model.Media, local *LocalMetadata) { if local.NSFW { m.NSFW = true } - if local.HasNFO || localHasDescriptiveMetadata(local) { + if local.HasNFO || (!local.PathHint && localHasDescriptiveMetadata(local)) { m.ScrapeStatus = "matched" } } @@ -2374,6 +2421,7 @@ func localHasDescriptiveMetadata(local *LocalMetadata) bool { local.Overview != "" || local.Rating > 0 || local.TMDbID > 0 || + local.BangumiID > 0 || local.DoubanID != "" || local.TheTVDBID != "" || local.Genres != "" || diff --git a/internal/service/scraper.go b/internal/service/scraper.go index ce5d39a..c901d86 100644 --- a/internal/service/scraper.go +++ b/internal/service/scraper.go @@ -239,6 +239,12 @@ func (s *ScraperService) EnrichOne(ctx context.Context, m *model.Media) error { } } + if match := s.matchFromMediaExternalIDs(ctx, m, lib); match != nil { + s.applyFanartArtwork(ctx, match) + mergeLocalMetadataIntoMatch(match, local) + return s.applyProviderMatch(ctx, m, lib, match) + } + candidates := scrapeQueryCandidates(m, lib) var query string match := (*Match)(nil) @@ -284,6 +290,43 @@ func (s *ScraperService) EnrichOne(ctx context.Context, m *model.Media) error { return s.applyProviderMatch(ctx, m, lib, match) } +func (s *ScraperService) matchFromMediaExternalIDs(ctx context.Context, m *model.Media, lib *model.Library) *Match { + if s == nil || m == nil { + return nil + } + mediaType := "" + if lib != nil { + mediaType = lib.Type + } + if m.TMDbID > 0 { + if match := s.manualTMDbMatchByID(ctx, m.TMDbID, normalizeMediaType(mediaType, m.Title, "")); match != nil { + return match + } + } + if strings.TrimSpace(m.DoubanID) != "" && s.douban != nil && s.douban.Enabled() { + if match, err := s.douban.GetMatchByID(ctx, strings.TrimSpace(m.DoubanID)); err == nil && match != nil { + return match + } else if err != nil { + s.log.Debug("douban id lookup failed", zap.String("media_id", m.ID), zap.String("douban_id", m.DoubanID), zap.Error(err)) + } + } + if m.BangumiID > 0 && s.bangumi != nil && s.bangumi.Enabled() { + if match, err := s.bangumi.GetSubject(ctx, m.BangumiID); err == nil && match != nil { + return match + } else if err != nil { + s.log.Debug("bangumi id lookup failed", zap.String("media_id", m.ID), zap.Int("bangumi_id", m.BangumiID), zap.Error(err)) + } + } + if strings.TrimSpace(m.TheTVDBID) != "" && s.thetvdb != nil && s.thetvdb.Enabled() { + if match, err := s.thetvdb.GetSeriesMatchByID(ctx, strings.TrimSpace(m.TheTVDBID)); err == nil && match != nil { + return match + } else if err != nil { + s.log.Debug("thetvdb id lookup failed", zap.String("media_id", m.ID), zap.String("thetvdb_id", m.TheTVDBID), zap.Error(err)) + } + } + return nil +} + func (s *ScraperService) applyFanartArtwork(ctx context.Context, match *Match) { if s == nil || s.fanart == nil || !s.fanart.Enabled() || match == nil { return diff --git a/internal/service/scraper_test.go b/internal/service/scraper_test.go index 547a62c..9d4e35b 100644 --- a/internal/service/scraper_test.go +++ b/internal/service/scraper_test.go @@ -45,6 +45,29 @@ func TestCleanQuery(t *testing.T) { } } +func TestExternalIDHintsFromText(t *testing.T) { + hints := externalIDHintsFromText("国漫/折腰 (2025) {tmdb 220269}/Season 1/折腰.S01E01.mkv") + if hints.TMDbID != 220269 { + t.Fatalf("tmdb hint = %d, want 220269", hints.TMDbID) + } + hints = externalIDHintsFromText("Movie (2026) {tmdb-1630433} [douban=3622222] {bgm 456789} {tvdb:12345}") + if hints.TMDbID != 1630433 || hints.DoubanID != "3622222" || hints.BangumiID != 456789 || hints.TheTVDBID != "12345" { + t.Fatalf("external hints not parsed: %+v", hints) + } +} + +func TestPathHintMetadataDoesNotMarkMediaMatched(t *testing.T) { + meta, hints := pathHintMetadata("cloud://openlist/国漫/折腰 (2025) {tmdb 220269}/Season 1/折腰.S01E01.mkv", true) + if meta == nil || hints.TMDbID != 220269 || meta.TMDbID != 220269 || meta.Title != "折腰" || meta.Year != 2025 { + t.Fatalf("path hint metadata = %+v hints=%+v", meta, hints) + } + media := &model.Media{Title: "折腰", ScrapeStatus: "pending"} + applyLocalMetadata(media, meta) + if media.ScrapeStatus != "pending" { + t.Fatalf("path hints alone must not mark media matched, got %q", media.ScrapeStatus) + } +} + func TestManualRequestMatchFallsBackToCandidatePayload(t *testing.T) { scraper := &ScraperService{} match, err := scraper.manualRequestMatch(t.Context(), ManualScrapeRequest{ @@ -134,6 +157,39 @@ func TestApplyManualMatchSavesSelectedCloudMatchWhenDetailsSlow(t *testing.T) { } } +func TestEnrichOneUsesExistingTMDbIDForCloudMedia(t *testing.T) { + scraper, repos, closeServer := newTestScraper(t) + defer closeServer() + + lib := model.Library{Name: "OpenList · 国漫", Path: "cloud://openlist/国漫", Type: "anime", Enabled: true} + if err := repos.DB.Create(&lib).Error; err != nil { + t.Fatal(err) + } + media := model.Media{ + LibraryID: lib.ID, + Title: "dirty release title", + Path: "cloud://openlist/国漫/间谍过家家 (2022) {tmdb-12345}/Season 1/间谍过家家.S01E01.2160p.mkv", + SeasonNum: 1, + EpisodeNum: 1, + TMDbID: 12345, + ScrapeStatus: "pending", + } + if err := repos.DB.Create(&media).Error; err != nil { + t.Fatal(err) + } + + if err := scraper.EnrichOne(t.Context(), &media); err != nil { + t.Fatal(err) + } + var got model.Media + if err := repos.DB.First(&got, "id = ?", media.ID).Error; err != nil { + t.Fatal(err) + } + if got.ScrapeStatus != "matched" || got.Title != "间谍过家家" || got.TMDbID != 12345 || got.PosterURL == "" { + t.Fatalf("tmdb id scrape did not apply match: title=%q status=%q tmdb=%d poster=%q", got.Title, got.ScrapeStatus, got.TMDbID, got.PosterURL) + } +} + func TestScrapeQueryCandidatesPreferSeriesFolderAndCJKTitle(t *testing.T) { lib := &model.Library{ Path: `F:\downloads\国产剧`, @@ -431,6 +487,13 @@ func newTestScraper(t *testing.T) (*ScraperService, *repository.Container, func( }) case strings.HasPrefix(r.URL.Path, "/tv/12345"): _ = json.NewEncoder(w).Encode(map[string]any{ + "id": 12345, + "name": "间谍过家家", + "overview": "测试简介", + "poster_path": "/poster.jpg", + "backdrop_path": "/backdrop.jpg", + "first_air_date": "2022-04-09", + "vote_average": 8.6, "origin_country": []string{"JP"}, "spoken_languages": []map[string]any{{ "iso_639_1": "ja", diff --git a/web/src/utils/groupSeries.ts b/web/src/utils/groupSeries.ts index 9c3b20f..3e78a8f 100644 --- a/web/src/utils/groupSeries.ts +++ b/web/src/utils/groupSeries.ts @@ -22,6 +22,10 @@ export function getSeriesKey(media: Media): string { if (media.series_id) return `series:${media.series_id}` const fromPath = seriesTitleFromPath(media.path) if (isEpisodeLike(media)) { + if (media.tmdb_id && media.tmdb_id > 0) return `tmdb:${media.tmdb_id}` + if (media.bangumi_id && media.bangumi_id > 0) return `bgm:${media.bangumi_id}` + if (media.douban_id) return `douban:${media.douban_id}` + if (media.thetvdb_id) return `thetvdb:${media.thetvdb_id}` return `lib:${media.library_id}|show:${normalizeTitle(fromPath || seriesTitle(media))}` } if (media.tmdb_id && media.tmdb_id > 0) return `tmdb:${media.tmdb_id}` @@ -44,6 +48,7 @@ function normalizeTitle(value?: string): string { .toLowerCase() .replace(/\s*\((?:19|20)\d{2}\)\s*/g, ' ') .replace(/\s*\[(?:tmdb|tmdbid)[=-]\d+\]\s*/g, ' ') + .replace(/\s*\{(?:tmdb|tmdbid|douban|bangumi|bgm|thetvdb|tvdb)[\s:=#-]*[a-z0-9_-]+\}\s*/g, ' ') .replace(/[\s._-]+/g, ' ') .trim() }