From f9dd082159a1f1c59d23825698454f2c8e3c3252 Mon Sep 17 00:00:00 2001 From: truewhile <779943132@qq.com> Date: Wed, 9 Sep 2026 17:36:11 +0800 Subject: [PATCH] =?UTF-8?q?=E4=BC=98=E5=8C=96=E5=8A=A8=E6=BC=AB=E5=88=AE?= =?UTF-8?q?=E5=89=8A?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- internal/service/external_id_hints.go | 7 + internal/service/media_classifier.go | 32 +- .../service/scraper_anime_theatrical_test.go | 283 ++++++++++++++++++ .../service/scraper_candidate_selection.go | 35 ++- internal/service/scraper_library.go | 15 +- internal/service/scraper_query.go | 104 +++++-- internal/service/scraper_query_paths.go | 23 +- 7 files changed, 453 insertions(+), 46 deletions(-) create mode 100644 internal/service/scraper_anime_theatrical_test.go diff --git a/internal/service/external_id_hints.go b/internal/service/external_id_hints.go index 7122585..f472246 100644 --- a/internal/service/external_id_hints.go +++ b/internal/service/external_id_hints.go @@ -78,6 +78,13 @@ func pathHintMetadata(raw string, seriesLike bool) (*LocalMetadata, mediaExterna title, year := "", 0 if seriesLike { title, year = CleanQuery(source) + } else if patTheatricalTitle.MatchString(raw) || patTheatricalFolder.MatchString(raw) { + base := pathBaseSlash(raw) + stem := mediaFileStem(base) + if stem == "" { + stem = strings.TrimSuffix(base, filepath.Ext(base)) + } + title, year = CleanQuery(stem) } else { title, year = cloudSeriesTitleFromMediaPath(source) if title == "" { diff --git a/internal/service/media_classifier.go b/internal/service/media_classifier.go index e9909bb..7195363 100644 --- a/internal/service/media_classifier.go +++ b/internal/service/media_classifier.go @@ -62,7 +62,7 @@ func classifyMediaCategory(input mediaClassifyInput, categories map[string]strin isWesternByCategory := containsAnyText(categoryText, "欧美剧", "欧美电视剧", "美剧", "英剧", "欧美电影", "外语电影") isWestern := isWesternByMetadata || (!hasRegionMetadata && isWesternByCategory) isUSAnime := hasAny(countries, "US") - hasAnimeText := containsAnyText(contentText, "动画", "动漫", "番剧", "年番", "国漫", "日番", "韩漫", "美漫", "bangumi", "anime", "b-global", "ani-one", "crunchyroll") + hasAnimeText := containsAnyText(contentText, "动画", "动漫", "番剧", "年番", "国漫", "日番", "韩漫", "美漫", "bangumi", "anime", "b-global", "ani-one", "crunchyroll", "剧场版", "劇場版", "动画电影", "動畫電影") hasVarietyText := containsAnyText(contentText, "综艺", "真人秀", "脱口秀", "晚会", "春晚", "gala", "festival gala", "reality", "talk show") hasDocumentaryText := containsAnyText(contentText, "纪录", "纪录片", "documentary", "docu", "national geographic", "natgeo") hasConcertText := containsAnyText(contentText, "演唱会", "音乐会", "concert", "live concert") @@ -191,21 +191,21 @@ func normalizeMediaType(mediaType, title, category string) string { return "adult" case containsAnyText(raw, "综艺", "真人秀"): return "variety" - case (containsAnyText(raw, "国漫", "日漫", "日番", "韩漫", "美漫", "欧美动漫", "其他动漫", "动漫", "动画") || classifierAnimeRE.MatchString(raw)) && !containsAnyText(raw, "动画电影"): - return "anime" - case containsAnyText(raw, "电视剧", "剧集", "连续剧", "短剧", "国产剧", "国剧", "大陆剧", "华语剧", "国产电视剧", "大陆电视剧", "华语电视剧", "欧美剧", "欧美电视剧", "美剧", "英剧", "日韩剧", "日韩电视剧", "日剧", "韩剧", "港剧", "台剧", "港台剧", "泰剧") || classifierTVRE.MatchString(raw): - return "tv" - case containsAnyText(raw, "电影", "演唱会") || classifierMovieRE.MatchString(raw): - return "movie" - } - text := strings.ToLower(title + " " + category) - switch { - case strings.Contains(text, "adult") || strings.Contains(text, "nsfw") || strings.Contains(text, "成人") || strings.Contains(text, "番号") || strings.Contains(text, "jav") || strings.Contains(text, "9kg") || classifierJAVCodeRE.MatchString(strings.ToUpper(title+" "+category)): - return "adult" - case containsAnyText(text, "综艺", "真人秀", "脱口秀", "晚会", "春晚", "gala", "festival gala", "reality", "talk show"): - return "variety" - case strings.Contains(text, "电影") || classifierMovieRE.MatchString(text): - return "movie" + case (containsAnyText(raw, "国漫", "日漫", "日番", "韩漫", "美漫", "欧美动漫", "其他动漫", "动漫", "动画") || classifierAnimeRE.MatchString(raw)) && !containsAnyText(raw, "动画电影", "剧场版", "劇場版"): + return "anime" + case containsAnyText(raw, "电视剧", "剧集", "连续剧", "短剧", "国产剧", "国剧", "大陆剧", "华语剧", "国产电视剧", "大陆电视剧", "华语电视剧", "欧美剧", "欧美电视剧", "美剧", "英剧", "日韩剧", "日韩电视剧", "日剧", "韩剧", "港剧", "台剧", "港台剧", "泰剧") || classifierTVRE.MatchString(raw): + return "tv" + case containsAnyText(raw, "电影", "演唱会", "剧场版", "劇場版") || classifierMovieRE.MatchString(raw): + return "movie" + } + text := strings.ToLower(title + " " + category) + switch { + case strings.Contains(text, "adult") || strings.Contains(text, "nsfw") || strings.Contains(text, "成人") || strings.Contains(text, "番号") || strings.Contains(text, "jav") || strings.Contains(text, "9kg") || classifierJAVCodeRE.MatchString(strings.ToUpper(title+" "+category)): + return "adult" + case containsAnyText(text, "综艺", "真人秀", "脱口秀", "晚会", "春晚", "gala", "festival gala", "reality", "talk show"): + return "variety" + case containsAnyText(text, "电影", "剧场版", "劇場版") || classifierMovieRE.MatchString(text): + return "movie" case classifierAnimeRE.MatchString(text) || strings.Contains(text, "动漫") || strings.Contains(text, "动画"): return "anime" case strings.Contains(text, "variety") || strings.Contains(text, "综艺") || strings.Contains(text, "真人秀"): diff --git a/internal/service/scraper_anime_theatrical_test.go b/internal/service/scraper_anime_theatrical_test.go new file mode 100644 index 0000000..8251b7a --- /dev/null +++ b/internal/service/scraper_anime_theatrical_test.go @@ -0,0 +1,283 @@ +package service + +import ( + "encoding/json" + "net/http" + "net/http/httptest" + "testing" + + "go.uber.org/zap" + + "github.com/truewhile/MeBox/internal/config" + "github.com/truewhile/MeBox/internal/model" +) + +func TestTheatricalTitleVariants(t *testing.T) { + tests := []struct { + input string + want []string + }{ + { + input: "名侦探柯南 剧场版01 引爆摩天楼", + want: []string{"名侦探柯南 引爆摩天楼", "名侦探柯南 剧场版01 引爆摩天楼"}, + }, + { + input: "名侦探柯南 剧场版26 黑铁的鱼影", + want: []string{"名侦探柯南 黑铁的鱼影", "名侦探柯南 剧场版26 黑铁的鱼影"}, + }, + { + input: "鬼灭之刃 剧场版 无限列车篇", + want: []string{"鬼灭之刃 无限列车篇", "鬼灭之刃 剧场版 无限列车篇"}, + }, + { + input: "航海王 The Movie 黄金之城", + want: []string{"航海王 The Movie 黄金之城"}, + }, + { + input: "普通动漫 第01集", + want: []string{"普通动漫 第01集"}, + }, + } + + for _, tc := range tests { + got := theatricalTitleVariants(tc.input) + if len(got) != len(tc.want) { + t.Fatalf("theatricalTitleVariants(%q) = %v, want %v", tc.input, got, tc.want) + } + for i := range got { + if got[i] != tc.want[i] { + t.Fatalf("theatricalTitleVariants(%q)[%d] = %q, want %q", tc.input, i, got[i], tc.want[i]) + } + } + } +} + +func TestMetadataMatchCompatibilityForTheatricalFeatures(t *testing.T) { + animeMatch := &Match{MediaType: "anime"} + + if metadataMatchCompatibleWithType("movie", animeMatch) { + t.Fatal("ordinary movie lookups must not accept anime matches") + } + if !metadataMatchCompatibleWithTheatrical("movie", true, animeMatch) { + t.Fatal("theatrical movie lookup should accept an anime provider match") + } + if metadataMatchCompatibleWithTheatrical("movie", false, animeMatch) { + t.Fatal("non-theatrical movie lookup must not accept an anime provider match") + } +} + +func TestScrapeQueryCandidatesForAnimeTheatricalMix(t *testing.T) { + lib := &model.Library{ + Path: `/media/anime`, + Type: "anime", + } + + // 1. 同目录下包含剧场版01 + theatricalMedia := &model.Media{ + Title: "名侦探柯南 剧场版01 引爆摩天楼", + Path: `/media/anime/名侦探柯南/名侦探柯南 剧场版01 引爆摩天楼.mkv`, + } + candidates := scrapeQueryCandidates(theatricalMedia, lib) + if len(candidates) == 0 { + t.Fatal("scrapeQueryCandidates returned no candidates for theatrical media") + } + if candidates[0] != "名侦探柯南 引爆摩天楼" { + t.Fatalf("first candidate for theatrical media = %q, want cleaned movie title '名侦探柯南 引爆摩天楼'; all=%v", candidates[0], candidates) + } + + // 2. 剧场版放在以剧场版命名的子目录 + subfolderTheatricalMedia := &model.Media{ + Title: "无限列车篇", + Path: `/media/anime/鬼灭之刃/剧场版/无限列车篇.mkv`, + } + subCandidates := scrapeQueryCandidates(subfolderTheatricalMedia, lib) + if len(subCandidates) == 0 { + t.Fatal("scrapeQueryCandidates returned no candidates for subfolder theatrical media") + } + foundCombined := false + for _, c := range subCandidates { + if c == "鬼灭之刃 无限列车篇" { + foundCombined = true + break + } + } + if !foundCombined { + t.Fatalf("candidates for subfolder theatrical did not contain '鬼灭之刃 无限列车篇': %v", subCandidates) + } + + // 3. 混合放置的普通剧集分集不应受剧场版影响 + episodeMedia := &model.Media{ + Title: "名侦探柯南 - S01E01", + Path: `/media/anime/名侦探柯南/名侦探柯南 - S01E01.mp4`, + SeasonNum: 1, + EpisodeNum: 1, + } + epCandidates := scrapeQueryCandidates(episodeMedia, lib) + if len(epCandidates) == 0 { + t.Fatal("scrapeQueryCandidates returned no candidates for episode media") + } + if epCandidates[0] != "名侦探柯南" { + t.Fatalf("first candidate for tv episode = %q, want series folder title '名侦探柯南'", epCandidates[0]) + } +} + +func TestEnrichOneAnimeMixedEpisodesAndTheatrical(t *testing.T) { + upstream := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + w.Header().Set("Content-Type", "application/json") + switch r.URL.Path { + case "/search/tv": + q := r.URL.Query().Get("query") + if q == "鬼灭之刃" { + _ = json.NewEncoder(w).Encode(map[string]any{ + "results": []map[string]any{{ + "id": 85937, + "name": "鬼灭之刃", + "original_name": "鬼滅の刃", + "overview": "大正时期、日本...", + "first_air_date": "2019-04-06", + }}, + }) + return + } + _ = json.NewEncoder(w).Encode(map[string]any{"results": []any{}}) + case "/search/movie": + q := r.URL.Query().Get("query") + if q == "鬼灭之刃 无限列车篇" || q == "鬼灭之刃 剧场版 无限列车篇" { + _ = json.NewEncoder(w).Encode(map[string]any{ + "results": []map[string]any{{ + "id": 635302, + "title": "鬼灭之刃 剧场版 无限列车篇", + "original_title": "劇場版「鬼滅の刃」無限列車編", + "overview": "在结束了蝴蝶屋的修业之后...", + "release_date": "2020-10-16", + }}, + }) + return + } + _ = json.NewEncoder(w).Encode(map[string]any{"results": []any{}}) + case "/tv/85937": + _ = json.NewEncoder(w).Encode(map[string]any{ + "id": 85937, + "name": "鬼灭之刃", + "original_name": "鬼滅の刃", + "first_air_date": "2019-04-06", + }) + case "/movie/635302": + _ = json.NewEncoder(w).Encode(map[string]any{ + "id": 635302, + "title": "鬼灭之刃 剧场版 无限列车篇", + "original_title": "劇場版「鬼滅の刃」無限列車編", + "release_date": "2020-10-16", + }) + default: + http.NotFound(w, r) + } + })) + defer upstream.Close() + + repos := newOrganizerTestRepo(t) + cfg := &config.Config{} + cfg.Secrets.TMDbAPIKey = "test-key" + cfg.Secrets.TMDbAPIProxy = upstream.URL + log := zap.NewNop() + scraper := NewScraperService(cfg, log, repos, NewTMDbProvider(cfg, log, nil), nil, nil, nil, NewHub(log)) + + lib := model.Library{Name: "动漫", Path: `/media/anime`, Type: "anime", Enabled: true} + if err := repos.DB.Create(&lib).Error; err != nil { + t.Fatal(err) + } + + // 1. 同一目录下的 TV 剧集 + tvEpisode := model.Media{ + LibraryID: lib.ID, + Title: "鬼灭之刃 S01E01", + Path: `/media/anime/鬼灭之刃/Season 01/鬼灭之刃 - S01E01.mkv`, + SeasonNum: 1, + EpisodeNum: 1, + ScrapeStatus: "pending", + } + if err := repos.DB.Create(&tvEpisode).Error; err != nil { + t.Fatal(err) + } + + // 2. 同一目录下的剧场版文件 + movieMedia := model.Media{ + LibraryID: lib.ID, + Title: "鬼灭之刃 剧场版 无限列车篇", + Path: `/media/anime/鬼灭之刃/鬼灭之刃 剧场版 无限列车篇.mkv`, + ScrapeStatus: "pending", + } + if err := repos.DB.Create(&movieMedia).Error; err != nil { + t.Fatal(err) + } + + // 刮削剧集 + if err := scraper.EnrichOne(t.Context(), &tvEpisode); err != nil { + t.Fatalf("enrich tv episode: %v", err) + } + gotTV, err := repos.Media.FindByID(t.Context(), tvEpisode.ID) + if err != nil || gotTV == nil { + t.Fatalf("load tv episode: %v", err) + } + if gotTV.ScrapeStatus != "matched" || gotTV.TMDbID != 85937 { + t.Fatalf("tv episode scrape mismatch: status=%s, tmdb_id=%d", gotTV.ScrapeStatus, gotTV.TMDbID) + } + + // 刮削剧场版 + if err := scraper.EnrichOne(t.Context(), &movieMedia); err != nil { + t.Fatalf("enrich movie: %v", err) + } + gotMovie, err := repos.Media.FindByID(t.Context(), movieMedia.ID) + if err != nil || gotMovie == nil { + t.Fatalf("load movie: %v", err) + } + if gotMovie.ScrapeStatus != "matched" || gotMovie.TMDbID != 635302 { + t.Fatalf("theatrical movie scrape mismatch: status=%s, tmdb_id=%d, want 635302", gotMovie.ScrapeStatus, gotMovie.TMDbID) + } + if gotMovie.Title != "鬼灭之刃 剧场版 无限列车篇" { + t.Fatalf("theatrical movie title = %q, want '鬼灭之刃 剧场版 无限列车篇'", gotMovie.Title) + } +} + +func TestClassifyAnimeTheatricalMovie(t *testing.T) { + categories := map[string]string{ + "animation_movie": "动画电影", + "jp_anime": "日番", + "chinese_movie": "华语电影", + } + tests := []struct { + title string + mediaType string + want string + }{ + { + title: "名侦探柯南 剧场版01 引爆摩天楼", + mediaType: "movie", + want: "动画电影", + }, + { + title: "鬼灭之刃 剧场版 无限列车篇", + mediaType: "movie", + want: "动画电影", + }, + { + title: "鬼灭之刃 S01E01", + mediaType: "tv", + want: "日番", + }, + } + + for _, tc := range tests { + input := mediaClassifyInput{ + MediaType: tc.mediaType, + Title: tc.title, + Languages: []string{"JA"}, + Countries: []string{"JP"}, + Genres: []string{"Animation"}, + } + got := classifyMediaCategory(input, categories) + if got != tc.want { + t.Fatalf("classifyMediaCategory(%q, %q) = %q, want %q", tc.title, tc.mediaType, got, tc.want) + } + } +} diff --git a/internal/service/scraper_candidate_selection.go b/internal/service/scraper_candidate_selection.go index ef0b5ee..dcd9f2b 100644 --- a/internal/service/scraper_candidate_selection.go +++ b/internal/service/scraper_candidate_selection.go @@ -8,6 +8,27 @@ import ( ) func (s *ScraperService) lookupAutomaticTMDb(ctx context.Context, kind, query string, year int) *Match { + if match := s.lookupAutomaticTMDbPrimary(ctx, kind, query, year); match != nil { + return match + } + fallbackKind := "" + normalized := normalizeOrganizeMediaType(kind) + if normalized == "anime" { + if isTVMetadataKind(kind) { + fallbackKind = "movie" + } else { + fallbackKind = "tv" + } + } + if fallbackKind != "" { + if fbMatch := s.lookupAutomaticTMDbPrimary(ctx, fallbackKind, query, year); fbMatch != nil { + return fbMatch + } + } + return nil +} + +func (s *ScraperService) lookupAutomaticTMDbPrimary(ctx context.Context, kind, query string, year int) *Match { var ( candidates []*Match err error @@ -175,8 +196,10 @@ func metadataMatchCompatibleWithType(expectedType string, match *Match) bool { return true } switch expectedType { - case "tv", "anime", "variety": + case "tv", "variety": return matchType == "tv" || matchType == "anime" || matchType == "variety" + case "anime": + return matchType == "anime" || matchType == "tv" || matchType == "movie" || matchType == "variety" case "movie", "adult": return matchType == "movie" || matchType == "adult" default: @@ -184,6 +207,16 @@ func metadataMatchCompatibleWithType(expectedType string, match *Match) bool { } } +func metadataMatchCompatibleWithTheatrical(expectedType string, isTheatrical bool, match *Match) bool { + if isTheatrical && + normalizeOrganizeMediaType(expectedType) == "movie" && + match != nil && + normalizeOrganizeMediaType(match.MediaType) == "anime" { + return true + } + return metadataMatchCompatibleWithType(expectedType, match) +} + func queryNeedsEnglishTMDbFallback(query string) bool { for _, r := range query { if (r >= 'a' && r <= 'z') || (r >= 'A' && r <= 'Z') { diff --git a/internal/service/scraper_library.go b/internal/service/scraper_library.go index 172b195..331775a 100644 --- a/internal/service/scraper_library.go +++ b/internal/service/scraper_library.go @@ -22,7 +22,10 @@ func (s *ScraperService) lookup(ctx context.Context, lib *model.Library, media * // was previously polluted into S01E20x. Do not let the dirty path override // that caller-supplied media type; TV/anime libraries still force TV below. explicitEpisode := media != nil && (media.SeasonNum > 0 || media.EpisodeNum > 0) - if (normalizeOrganizeMediaType(kind) != "movie" || explicitEpisode) && mediaIsEpisodic(media, lib) { + isTheatrical := mediaLooksLikeTheatricalFeature(media) + if isTheatrical { + kind = "movie" + } else if (normalizeOrganizeMediaType(kind) != "movie" || explicitEpisode) && mediaIsEpisodic(media, lib) { kind = "tv" } if s.tmdb != nil && s.tmdb.Enabled() { @@ -30,9 +33,15 @@ func (s *ScraperService) lookup(ctx context.Context, lib *model.Library, media * match.Provider = "tmdb" return match } + if isTheatrical { + if match := s.lookupAutomaticTMDb(ctx, "tv", query, year); match != nil { + match.Provider = "tmdb" + return match + } + } } if s.douban != nil && s.douban.Enabled() { - if m, err := s.douban.SearchMatch(ctx, query); err == nil && m != nil && metadataMatchCompatibleWithType(kind, m) { + if m, err := s.douban.SearchMatch(ctx, query); err == nil && m != nil && metadataMatchCompatibleWithTheatrical(kind, isTheatrical, m) { m.Provider = "douban" return m } else if err != nil { @@ -40,7 +49,7 @@ func (s *ScraperService) lookup(ctx context.Context, lib *model.Library, media * } } if s.bangumi != nil && s.bangumi.Enabled() { - if m, err := s.bangumi.Search(ctx, query); err == nil && m != nil && metadataMatchCompatibleWithType(kind, m) { + if m, err := s.bangumi.Search(ctx, query); err == nil && m != nil && metadataMatchCompatibleWithTheatrical(kind, isTheatrical, m) { m.Provider = "bangumi" return m } else if err != nil { diff --git a/internal/service/scraper_query.go b/internal/service/scraper_query.go index cb07ad0..4ed2bab 100644 --- a/internal/service/scraper_query.go +++ b/internal/service/scraper_query.go @@ -15,8 +15,41 @@ var ( episodeTitleQueryRE = regexp.MustCompile(`^\s*第\s*[0-9一二三四五六七八九十百零两]+\s*[集期话話](?:\s*[上下])?\s*[::].+`) genericEpisodeWordsRE = regexp.MustCompile(`^\s*第\s*[集期话話]\s*$`) episodeReleaseTitleTagRE = regexp.MustCompile(`(?i)(?:^|[\s._-])s\d{1,2}e\d{1,3}(?:[\s._-]|$)`) + patTheatricalTitle = regexp.MustCompile(`(?i)(?:剧场版|劇場版|动画电影|動畫電影|电影版|電影版|\bthe\s+movie\b|\bmovie\s*\d{1,2}\b)`) + patTheatricalFolder = regexp.MustCompile(`(?i)[\\/](?:剧场版|劇場版|動畫電影|动画电影)[\\/]`) + theatricalNoiseRE = regexp.MustCompile(`(?i)(?:剧场版|劇場版)\s*(?:第?\s*\d{1,3}\s*[部篇]?)?|电影版|電影版|动画电影|動畫電影`) ) +func theatricalTitleVariants(raw string) []string { + raw = strings.TrimSpace(raw) + if raw == "" { + return nil + } + if !theatricalNoiseRE.MatchString(raw) { + return []string{raw} + } + stripped := theatricalNoiseRE.ReplaceAllString(raw, " ") + stripped = strings.Join(strings.Fields(stripped), " ") + if stripped != "" && !strings.EqualFold(stripped, raw) { + return []string{stripped, raw} + } + return []string{raw} +} + +func mediaLooksLikeTheatricalFeature(m *model.Media) bool { + if m == nil { + return false + } + if m.SeasonNum > 0 || m.EpisodeNum > 0 { + return false + } + if season, ep := ParseEpisode(m.Path); season > 0 || ep > 0 { + return false + } + text := m.Title + " " + pathBaseSlash(m.Path) + return patTheatricalTitle.MatchString(text) || patTheatricalFolder.MatchString(m.Path) +} + func scrapeQueryCandidates(m *model.Media, lib *model.Library) []string { return scrapeQueryCandidatesWithNormalizer(m, lib, func(raw string) (string, int) { return CleanQuery(raw) @@ -33,31 +66,59 @@ func scrapeQueryCandidatesWithNormalizer(m *model.Media, lib *model.Library, cle seen := map[string]struct{}{} var out []string add := func(raw string) { - cleaned, _ := clean(raw) - if cleaned == "" { - cleaned = strings.TrimSpace(raw) - } - for _, candidate := range titleCandidates(cleaned) { - if unsafeAutomaticEpisodeQuery(candidate) { - continue + for _, variant := range theatricalTitleVariants(raw) { + cleaned, _ := clean(variant) + if cleaned == "" { + cleaned = strings.TrimSpace(variant) } - key := strings.ToLower(candidate) - if _, ok := seen[key]; ok || candidate == "" { - continue + for _, candidate := range titleCandidates(cleaned) { + if unsafeAutomaticEpisodeQuery(candidate) { + continue + } + key := strings.ToLower(candidate) + if _, ok := seen[key]; ok || candidate == "" { + continue + } + seen[key] = struct{}{} + out = append(out, candidate) } - seen[key] = struct{}{} - out = append(out, candidate) } } - episodic := mediaIsEpisodic(m, lib) - if lib != nil && episodic { - add(seriesFolderTitle(m.Path, lib.Path)) + + isTheatrical := mediaLooksLikeTheatricalFeature(m) + if isTheatrical { + add(m.Title) + add(m.Path) + if lib != nil { + seriesTitle := seriesFolderTitle(m.Path, lib.Path) + if seriesTitle != "" { + baseName := pathBaseSlash(m.Path) + stem := mediaFileStem(baseName) + if stem == "" { + stem = strings.TrimSuffix(baseName, filepath.Ext(baseName)) + } + cleanStem, _ := clean(stem) + if cleanStem == "" { + cleanStem = stem + } + if !strings.Contains(strings.ToLower(cleanStem), strings.ToLower(seriesTitle)) { + add(seriesTitle + " " + cleanStem) + } + } + add(mediaFolderTitle(m.Path, lib.Path)) + } + } else { + episodic := mediaIsEpisodic(m, lib) + if lib != nil && episodic { + add(seriesFolderTitle(m.Path, lib.Path)) + } + if lib != nil { + add(mediaFolderTitle(m.Path, lib.Path)) + } + add(m.Title) + add(m.Path) } - if lib != nil { - add(mediaFolderTitle(m.Path, lib.Path)) - } - add(m.Title) - add(m.Path) + if len(out) == 0 { base := pathBaseSlash(m.Path) out = append(out, strings.TrimSuffix(base, filepath.Ext(base))) @@ -106,6 +167,9 @@ func containsCJK(s string) bool { } func mediaIsEpisodic(m *model.Media, lib *model.Library) bool { + if m != nil && mediaLooksLikeTheatricalFeature(m) { + return false + } if m != nil && (m.SeasonNum > 0 || m.EpisodeNum > 0) { return true } diff --git a/internal/service/scraper_query_paths.go b/internal/service/scraper_query_paths.go index 236a6ef..7053438 100644 --- a/internal/service/scraper_query_paths.go +++ b/internal/service/scraper_query_paths.go @@ -13,10 +13,10 @@ func mediaFolderTitle(mediaPath, libraryRoot string) string { if base == "" || base == "." { return "" } - if isTechnicalMediaFolder(base) || strictSeasonFolderMatched(base) { - dir = parentSlashPath(dir) - continue - } + if isTechnicalMediaFolder(base) || strictSeasonFolderMatched(base) || isTheatricalFolder(base) { + dir = parentSlashPath(dir) + continue + } if isMediaCollectionFolder(base) { dir = parentSlashPath(dir) continue @@ -104,7 +104,7 @@ func parentSlashPath(value string) string { func seriesFolderTitle(mediaPath, libraryRoot string) string { dir := parentSlashPath(mediaPath) - if strictSeasonFolderMatched(pathBaseSlash(dir)) { + if strictSeasonFolderMatched(pathBaseSlash(dir)) || isTheatricalFolder(pathBaseSlash(dir)) { dir = parentSlashPath(dir) } if root := comparableLibraryRoot(libraryRoot); root != "" && sameSlashPath(dir, root) { @@ -114,12 +114,23 @@ func seriesFolderTitle(mediaPath, libraryRoot string) string { if base == "" || base == "." { return "" } - if isGenericMediaCategoryFolder(base) || isTechnicalMediaFolder(base) || strictSeasonFolderMatched(base) { + if isGenericMediaCategoryFolder(base) || isTechnicalMediaFolder(base) || strictSeasonFolderMatched(base) || isTheatricalFolder(base) { return "" } return base } +func isTheatricalFolder(name string) bool { + key := strings.ToLower(strings.TrimSpace(name)) + key = strings.Trim(key, `\/`) + switch key { + case "剧场版", "劇場版", "动画电影", "動畫電影", "特别篇", "特別篇", "specials", "sp", "ova", "oad": + return true + default: + return false + } +} + func libraryRootTitle(libraryRoot string) string { base := "" if info, ok := ParseCloudLibraryMount(libraryRoot); ok {