mirror of
https://github.com/truewhile/MeBox.git
synced 2026-09-30 03:36:37 +08:00
优化动漫刮削
This commit is contained in:
@@ -78,6 +78,13 @@ func pathHintMetadata(raw string, seriesLike bool) (*LocalMetadata, mediaExterna
|
||||
title, year := "", 0
|
||||
if seriesLike {
|
||||
title, year = CleanQuery(source)
|
||||
} else if patTheatricalTitle.MatchString(raw) || patTheatricalFolder.MatchString(raw) {
|
||||
base := pathBaseSlash(raw)
|
||||
stem := mediaFileStem(base)
|
||||
if stem == "" {
|
||||
stem = strings.TrimSuffix(base, filepath.Ext(base))
|
||||
}
|
||||
title, year = CleanQuery(stem)
|
||||
} else {
|
||||
title, year = cloudSeriesTitleFromMediaPath(source)
|
||||
if title == "" {
|
||||
|
||||
@@ -62,7 +62,7 @@ func classifyMediaCategory(input mediaClassifyInput, categories map[string]strin
|
||||
isWesternByCategory := containsAnyText(categoryText, "欧美剧", "欧美电视剧", "美剧", "英剧", "欧美电影", "外语电影")
|
||||
isWestern := isWesternByMetadata || (!hasRegionMetadata && isWesternByCategory)
|
||||
isUSAnime := hasAny(countries, "US")
|
||||
hasAnimeText := containsAnyText(contentText, "动画", "动漫", "番剧", "年番", "国漫", "日番", "韩漫", "美漫", "bangumi", "anime", "b-global", "ani-one", "crunchyroll")
|
||||
hasAnimeText := containsAnyText(contentText, "动画", "动漫", "番剧", "年番", "国漫", "日番", "韩漫", "美漫", "bangumi", "anime", "b-global", "ani-one", "crunchyroll", "剧场版", "劇場版", "动画电影", "動畫電影")
|
||||
hasVarietyText := containsAnyText(contentText, "综艺", "真人秀", "脱口秀", "晚会", "春晚", "gala", "festival gala", "reality", "talk show")
|
||||
hasDocumentaryText := containsAnyText(contentText, "纪录", "纪录片", "documentary", "docu", "national geographic", "natgeo")
|
||||
hasConcertText := containsAnyText(contentText, "演唱会", "音乐会", "concert", "live concert")
|
||||
@@ -191,21 +191,21 @@ func normalizeMediaType(mediaType, title, category string) string {
|
||||
return "adult"
|
||||
case containsAnyText(raw, "综艺", "真人秀"):
|
||||
return "variety"
|
||||
case (containsAnyText(raw, "国漫", "日漫", "日番", "韩漫", "美漫", "欧美动漫", "其他动漫", "动漫", "动画") || classifierAnimeRE.MatchString(raw)) && !containsAnyText(raw, "动画电影"):
|
||||
return "anime"
|
||||
case containsAnyText(raw, "电视剧", "剧集", "连续剧", "短剧", "国产剧", "国剧", "大陆剧", "华语剧", "国产电视剧", "大陆电视剧", "华语电视剧", "欧美剧", "欧美电视剧", "美剧", "英剧", "日韩剧", "日韩电视剧", "日剧", "韩剧", "港剧", "台剧", "港台剧", "泰剧") || classifierTVRE.MatchString(raw):
|
||||
return "tv"
|
||||
case containsAnyText(raw, "电影", "演唱会") || classifierMovieRE.MatchString(raw):
|
||||
return "movie"
|
||||
}
|
||||
text := strings.ToLower(title + " " + category)
|
||||
switch {
|
||||
case strings.Contains(text, "adult") || strings.Contains(text, "nsfw") || strings.Contains(text, "成人") || strings.Contains(text, "番号") || strings.Contains(text, "jav") || strings.Contains(text, "9kg") || classifierJAVCodeRE.MatchString(strings.ToUpper(title+" "+category)):
|
||||
return "adult"
|
||||
case containsAnyText(text, "综艺", "真人秀", "脱口秀", "晚会", "春晚", "gala", "festival gala", "reality", "talk show"):
|
||||
return "variety"
|
||||
case strings.Contains(text, "电影") || classifierMovieRE.MatchString(text):
|
||||
return "movie"
|
||||
case (containsAnyText(raw, "国漫", "日漫", "日番", "韩漫", "美漫", "欧美动漫", "其他动漫", "动漫", "动画") || classifierAnimeRE.MatchString(raw)) && !containsAnyText(raw, "动画电影", "剧场版", "劇場版"):
|
||||
return "anime"
|
||||
case containsAnyText(raw, "电视剧", "剧集", "连续剧", "短剧", "国产剧", "国剧", "大陆剧", "华语剧", "国产电视剧", "大陆电视剧", "华语电视剧", "欧美剧", "欧美电视剧", "美剧", "英剧", "日韩剧", "日韩电视剧", "日剧", "韩剧", "港剧", "台剧", "港台剧", "泰剧") || classifierTVRE.MatchString(raw):
|
||||
return "tv"
|
||||
case containsAnyText(raw, "电影", "演唱会", "剧场版", "劇場版") || classifierMovieRE.MatchString(raw):
|
||||
return "movie"
|
||||
}
|
||||
text := strings.ToLower(title + " " + category)
|
||||
switch {
|
||||
case strings.Contains(text, "adult") || strings.Contains(text, "nsfw") || strings.Contains(text, "成人") || strings.Contains(text, "番号") || strings.Contains(text, "jav") || strings.Contains(text, "9kg") || classifierJAVCodeRE.MatchString(strings.ToUpper(title+" "+category)):
|
||||
return "adult"
|
||||
case containsAnyText(text, "综艺", "真人秀", "脱口秀", "晚会", "春晚", "gala", "festival gala", "reality", "talk show"):
|
||||
return "variety"
|
||||
case containsAnyText(text, "电影", "剧场版", "劇場版") || classifierMovieRE.MatchString(text):
|
||||
return "movie"
|
||||
case classifierAnimeRE.MatchString(text) || strings.Contains(text, "动漫") || strings.Contains(text, "动画"):
|
||||
return "anime"
|
||||
case strings.Contains(text, "variety") || strings.Contains(text, "综艺") || strings.Contains(text, "真人秀"):
|
||||
|
||||
@@ -0,0 +1,283 @@
|
||||
package service
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"testing"
|
||||
|
||||
"go.uber.org/zap"
|
||||
|
||||
"github.com/truewhile/MeBox/internal/config"
|
||||
"github.com/truewhile/MeBox/internal/model"
|
||||
)
|
||||
|
||||
func TestTheatricalTitleVariants(t *testing.T) {
|
||||
tests := []struct {
|
||||
input string
|
||||
want []string
|
||||
}{
|
||||
{
|
||||
input: "名侦探柯南 剧场版01 引爆摩天楼",
|
||||
want: []string{"名侦探柯南 引爆摩天楼", "名侦探柯南 剧场版01 引爆摩天楼"},
|
||||
},
|
||||
{
|
||||
input: "名侦探柯南 剧场版26 黑铁的鱼影",
|
||||
want: []string{"名侦探柯南 黑铁的鱼影", "名侦探柯南 剧场版26 黑铁的鱼影"},
|
||||
},
|
||||
{
|
||||
input: "鬼灭之刃 剧场版 无限列车篇",
|
||||
want: []string{"鬼灭之刃 无限列车篇", "鬼灭之刃 剧场版 无限列车篇"},
|
||||
},
|
||||
{
|
||||
input: "航海王 The Movie 黄金之城",
|
||||
want: []string{"航海王 The Movie 黄金之城"},
|
||||
},
|
||||
{
|
||||
input: "普通动漫 第01集",
|
||||
want: []string{"普通动漫 第01集"},
|
||||
},
|
||||
}
|
||||
|
||||
for _, tc := range tests {
|
||||
got := theatricalTitleVariants(tc.input)
|
||||
if len(got) != len(tc.want) {
|
||||
t.Fatalf("theatricalTitleVariants(%q) = %v, want %v", tc.input, got, tc.want)
|
||||
}
|
||||
for i := range got {
|
||||
if got[i] != tc.want[i] {
|
||||
t.Fatalf("theatricalTitleVariants(%q)[%d] = %q, want %q", tc.input, i, got[i], tc.want[i])
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestMetadataMatchCompatibilityForTheatricalFeatures(t *testing.T) {
|
||||
animeMatch := &Match{MediaType: "anime"}
|
||||
|
||||
if metadataMatchCompatibleWithType("movie", animeMatch) {
|
||||
t.Fatal("ordinary movie lookups must not accept anime matches")
|
||||
}
|
||||
if !metadataMatchCompatibleWithTheatrical("movie", true, animeMatch) {
|
||||
t.Fatal("theatrical movie lookup should accept an anime provider match")
|
||||
}
|
||||
if metadataMatchCompatibleWithTheatrical("movie", false, animeMatch) {
|
||||
t.Fatal("non-theatrical movie lookup must not accept an anime provider match")
|
||||
}
|
||||
}
|
||||
|
||||
func TestScrapeQueryCandidatesForAnimeTheatricalMix(t *testing.T) {
|
||||
lib := &model.Library{
|
||||
Path: `/media/anime`,
|
||||
Type: "anime",
|
||||
}
|
||||
|
||||
// 1. 同目录下包含剧场版01
|
||||
theatricalMedia := &model.Media{
|
||||
Title: "名侦探柯南 剧场版01 引爆摩天楼",
|
||||
Path: `/media/anime/名侦探柯南/名侦探柯南 剧场版01 引爆摩天楼.mkv`,
|
||||
}
|
||||
candidates := scrapeQueryCandidates(theatricalMedia, lib)
|
||||
if len(candidates) == 0 {
|
||||
t.Fatal("scrapeQueryCandidates returned no candidates for theatrical media")
|
||||
}
|
||||
if candidates[0] != "名侦探柯南 引爆摩天楼" {
|
||||
t.Fatalf("first candidate for theatrical media = %q, want cleaned movie title '名侦探柯南 引爆摩天楼'; all=%v", candidates[0], candidates)
|
||||
}
|
||||
|
||||
// 2. 剧场版放在以剧场版命名的子目录
|
||||
subfolderTheatricalMedia := &model.Media{
|
||||
Title: "无限列车篇",
|
||||
Path: `/media/anime/鬼灭之刃/剧场版/无限列车篇.mkv`,
|
||||
}
|
||||
subCandidates := scrapeQueryCandidates(subfolderTheatricalMedia, lib)
|
||||
if len(subCandidates) == 0 {
|
||||
t.Fatal("scrapeQueryCandidates returned no candidates for subfolder theatrical media")
|
||||
}
|
||||
foundCombined := false
|
||||
for _, c := range subCandidates {
|
||||
if c == "鬼灭之刃 无限列车篇" {
|
||||
foundCombined = true
|
||||
break
|
||||
}
|
||||
}
|
||||
if !foundCombined {
|
||||
t.Fatalf("candidates for subfolder theatrical did not contain '鬼灭之刃 无限列车篇': %v", subCandidates)
|
||||
}
|
||||
|
||||
// 3. 混合放置的普通剧集分集不应受剧场版影响
|
||||
episodeMedia := &model.Media{
|
||||
Title: "名侦探柯南 - S01E01",
|
||||
Path: `/media/anime/名侦探柯南/名侦探柯南 - S01E01.mp4`,
|
||||
SeasonNum: 1,
|
||||
EpisodeNum: 1,
|
||||
}
|
||||
epCandidates := scrapeQueryCandidates(episodeMedia, lib)
|
||||
if len(epCandidates) == 0 {
|
||||
t.Fatal("scrapeQueryCandidates returned no candidates for episode media")
|
||||
}
|
||||
if epCandidates[0] != "名侦探柯南" {
|
||||
t.Fatalf("first candidate for tv episode = %q, want series folder title '名侦探柯南'", epCandidates[0])
|
||||
}
|
||||
}
|
||||
|
||||
func TestEnrichOneAnimeMixedEpisodesAndTheatrical(t *testing.T) {
|
||||
upstream := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
w.Header().Set("Content-Type", "application/json")
|
||||
switch r.URL.Path {
|
||||
case "/search/tv":
|
||||
q := r.URL.Query().Get("query")
|
||||
if q == "鬼灭之刃" {
|
||||
_ = json.NewEncoder(w).Encode(map[string]any{
|
||||
"results": []map[string]any{{
|
||||
"id": 85937,
|
||||
"name": "鬼灭之刃",
|
||||
"original_name": "鬼滅の刃",
|
||||
"overview": "大正时期、日本...",
|
||||
"first_air_date": "2019-04-06",
|
||||
}},
|
||||
})
|
||||
return
|
||||
}
|
||||
_ = json.NewEncoder(w).Encode(map[string]any{"results": []any{}})
|
||||
case "/search/movie":
|
||||
q := r.URL.Query().Get("query")
|
||||
if q == "鬼灭之刃 无限列车篇" || q == "鬼灭之刃 剧场版 无限列车篇" {
|
||||
_ = json.NewEncoder(w).Encode(map[string]any{
|
||||
"results": []map[string]any{{
|
||||
"id": 635302,
|
||||
"title": "鬼灭之刃 剧场版 无限列车篇",
|
||||
"original_title": "劇場版「鬼滅の刃」無限列車編",
|
||||
"overview": "在结束了蝴蝶屋的修业之后...",
|
||||
"release_date": "2020-10-16",
|
||||
}},
|
||||
})
|
||||
return
|
||||
}
|
||||
_ = json.NewEncoder(w).Encode(map[string]any{"results": []any{}})
|
||||
case "/tv/85937":
|
||||
_ = json.NewEncoder(w).Encode(map[string]any{
|
||||
"id": 85937,
|
||||
"name": "鬼灭之刃",
|
||||
"original_name": "鬼滅の刃",
|
||||
"first_air_date": "2019-04-06",
|
||||
})
|
||||
case "/movie/635302":
|
||||
_ = json.NewEncoder(w).Encode(map[string]any{
|
||||
"id": 635302,
|
||||
"title": "鬼灭之刃 剧场版 无限列车篇",
|
||||
"original_title": "劇場版「鬼滅の刃」無限列車編",
|
||||
"release_date": "2020-10-16",
|
||||
})
|
||||
default:
|
||||
http.NotFound(w, r)
|
||||
}
|
||||
}))
|
||||
defer upstream.Close()
|
||||
|
||||
repos := newOrganizerTestRepo(t)
|
||||
cfg := &config.Config{}
|
||||
cfg.Secrets.TMDbAPIKey = "test-key"
|
||||
cfg.Secrets.TMDbAPIProxy = upstream.URL
|
||||
log := zap.NewNop()
|
||||
scraper := NewScraperService(cfg, log, repos, NewTMDbProvider(cfg, log, nil), nil, nil, nil, NewHub(log))
|
||||
|
||||
lib := model.Library{Name: "动漫", Path: `/media/anime`, Type: "anime", Enabled: true}
|
||||
if err := repos.DB.Create(&lib).Error; err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
// 1. 同一目录下的 TV 剧集
|
||||
tvEpisode := model.Media{
|
||||
LibraryID: lib.ID,
|
||||
Title: "鬼灭之刃 S01E01",
|
||||
Path: `/media/anime/鬼灭之刃/Season 01/鬼灭之刃 - S01E01.mkv`,
|
||||
SeasonNum: 1,
|
||||
EpisodeNum: 1,
|
||||
ScrapeStatus: "pending",
|
||||
}
|
||||
if err := repos.DB.Create(&tvEpisode).Error; err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
// 2. 同一目录下的剧场版文件
|
||||
movieMedia := model.Media{
|
||||
LibraryID: lib.ID,
|
||||
Title: "鬼灭之刃 剧场版 无限列车篇",
|
||||
Path: `/media/anime/鬼灭之刃/鬼灭之刃 剧场版 无限列车篇.mkv`,
|
||||
ScrapeStatus: "pending",
|
||||
}
|
||||
if err := repos.DB.Create(&movieMedia).Error; err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
// 刮削剧集
|
||||
if err := scraper.EnrichOne(t.Context(), &tvEpisode); err != nil {
|
||||
t.Fatalf("enrich tv episode: %v", err)
|
||||
}
|
||||
gotTV, err := repos.Media.FindByID(t.Context(), tvEpisode.ID)
|
||||
if err != nil || gotTV == nil {
|
||||
t.Fatalf("load tv episode: %v", err)
|
||||
}
|
||||
if gotTV.ScrapeStatus != "matched" || gotTV.TMDbID != 85937 {
|
||||
t.Fatalf("tv episode scrape mismatch: status=%s, tmdb_id=%d", gotTV.ScrapeStatus, gotTV.TMDbID)
|
||||
}
|
||||
|
||||
// 刮削剧场版
|
||||
if err := scraper.EnrichOne(t.Context(), &movieMedia); err != nil {
|
||||
t.Fatalf("enrich movie: %v", err)
|
||||
}
|
||||
gotMovie, err := repos.Media.FindByID(t.Context(), movieMedia.ID)
|
||||
if err != nil || gotMovie == nil {
|
||||
t.Fatalf("load movie: %v", err)
|
||||
}
|
||||
if gotMovie.ScrapeStatus != "matched" || gotMovie.TMDbID != 635302 {
|
||||
t.Fatalf("theatrical movie scrape mismatch: status=%s, tmdb_id=%d, want 635302", gotMovie.ScrapeStatus, gotMovie.TMDbID)
|
||||
}
|
||||
if gotMovie.Title != "鬼灭之刃 剧场版 无限列车篇" {
|
||||
t.Fatalf("theatrical movie title = %q, want '鬼灭之刃 剧场版 无限列车篇'", gotMovie.Title)
|
||||
}
|
||||
}
|
||||
|
||||
func TestClassifyAnimeTheatricalMovie(t *testing.T) {
|
||||
categories := map[string]string{
|
||||
"animation_movie": "动画电影",
|
||||
"jp_anime": "日番",
|
||||
"chinese_movie": "华语电影",
|
||||
}
|
||||
tests := []struct {
|
||||
title string
|
||||
mediaType string
|
||||
want string
|
||||
}{
|
||||
{
|
||||
title: "名侦探柯南 剧场版01 引爆摩天楼",
|
||||
mediaType: "movie",
|
||||
want: "动画电影",
|
||||
},
|
||||
{
|
||||
title: "鬼灭之刃 剧场版 无限列车篇",
|
||||
mediaType: "movie",
|
||||
want: "动画电影",
|
||||
},
|
||||
{
|
||||
title: "鬼灭之刃 S01E01",
|
||||
mediaType: "tv",
|
||||
want: "日番",
|
||||
},
|
||||
}
|
||||
|
||||
for _, tc := range tests {
|
||||
input := mediaClassifyInput{
|
||||
MediaType: tc.mediaType,
|
||||
Title: tc.title,
|
||||
Languages: []string{"JA"},
|
||||
Countries: []string{"JP"},
|
||||
Genres: []string{"Animation"},
|
||||
}
|
||||
got := classifyMediaCategory(input, categories)
|
||||
if got != tc.want {
|
||||
t.Fatalf("classifyMediaCategory(%q, %q) = %q, want %q", tc.title, tc.mediaType, got, tc.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -8,6 +8,27 @@ import (
|
||||
)
|
||||
|
||||
func (s *ScraperService) lookupAutomaticTMDb(ctx context.Context, kind, query string, year int) *Match {
|
||||
if match := s.lookupAutomaticTMDbPrimary(ctx, kind, query, year); match != nil {
|
||||
return match
|
||||
}
|
||||
fallbackKind := ""
|
||||
normalized := normalizeOrganizeMediaType(kind)
|
||||
if normalized == "anime" {
|
||||
if isTVMetadataKind(kind) {
|
||||
fallbackKind = "movie"
|
||||
} else {
|
||||
fallbackKind = "tv"
|
||||
}
|
||||
}
|
||||
if fallbackKind != "" {
|
||||
if fbMatch := s.lookupAutomaticTMDbPrimary(ctx, fallbackKind, query, year); fbMatch != nil {
|
||||
return fbMatch
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func (s *ScraperService) lookupAutomaticTMDbPrimary(ctx context.Context, kind, query string, year int) *Match {
|
||||
var (
|
||||
candidates []*Match
|
||||
err error
|
||||
@@ -175,8 +196,10 @@ func metadataMatchCompatibleWithType(expectedType string, match *Match) bool {
|
||||
return true
|
||||
}
|
||||
switch expectedType {
|
||||
case "tv", "anime", "variety":
|
||||
case "tv", "variety":
|
||||
return matchType == "tv" || matchType == "anime" || matchType == "variety"
|
||||
case "anime":
|
||||
return matchType == "anime" || matchType == "tv" || matchType == "movie" || matchType == "variety"
|
||||
case "movie", "adult":
|
||||
return matchType == "movie" || matchType == "adult"
|
||||
default:
|
||||
@@ -184,6 +207,16 @@ func metadataMatchCompatibleWithType(expectedType string, match *Match) bool {
|
||||
}
|
||||
}
|
||||
|
||||
func metadataMatchCompatibleWithTheatrical(expectedType string, isTheatrical bool, match *Match) bool {
|
||||
if isTheatrical &&
|
||||
normalizeOrganizeMediaType(expectedType) == "movie" &&
|
||||
match != nil &&
|
||||
normalizeOrganizeMediaType(match.MediaType) == "anime" {
|
||||
return true
|
||||
}
|
||||
return metadataMatchCompatibleWithType(expectedType, match)
|
||||
}
|
||||
|
||||
func queryNeedsEnglishTMDbFallback(query string) bool {
|
||||
for _, r := range query {
|
||||
if (r >= 'a' && r <= 'z') || (r >= 'A' && r <= 'Z') {
|
||||
|
||||
@@ -22,7 +22,10 @@ func (s *ScraperService) lookup(ctx context.Context, lib *model.Library, media *
|
||||
// was previously polluted into S01E20x. Do not let the dirty path override
|
||||
// that caller-supplied media type; TV/anime libraries still force TV below.
|
||||
explicitEpisode := media != nil && (media.SeasonNum > 0 || media.EpisodeNum > 0)
|
||||
if (normalizeOrganizeMediaType(kind) != "movie" || explicitEpisode) && mediaIsEpisodic(media, lib) {
|
||||
isTheatrical := mediaLooksLikeTheatricalFeature(media)
|
||||
if isTheatrical {
|
||||
kind = "movie"
|
||||
} else if (normalizeOrganizeMediaType(kind) != "movie" || explicitEpisode) && mediaIsEpisodic(media, lib) {
|
||||
kind = "tv"
|
||||
}
|
||||
if s.tmdb != nil && s.tmdb.Enabled() {
|
||||
@@ -30,9 +33,15 @@ func (s *ScraperService) lookup(ctx context.Context, lib *model.Library, media *
|
||||
match.Provider = "tmdb"
|
||||
return match
|
||||
}
|
||||
if isTheatrical {
|
||||
if match := s.lookupAutomaticTMDb(ctx, "tv", query, year); match != nil {
|
||||
match.Provider = "tmdb"
|
||||
return match
|
||||
}
|
||||
}
|
||||
}
|
||||
if s.douban != nil && s.douban.Enabled() {
|
||||
if m, err := s.douban.SearchMatch(ctx, query); err == nil && m != nil && metadataMatchCompatibleWithType(kind, m) {
|
||||
if m, err := s.douban.SearchMatch(ctx, query); err == nil && m != nil && metadataMatchCompatibleWithTheatrical(kind, isTheatrical, m) {
|
||||
m.Provider = "douban"
|
||||
return m
|
||||
} else if err != nil {
|
||||
@@ -40,7 +49,7 @@ func (s *ScraperService) lookup(ctx context.Context, lib *model.Library, media *
|
||||
}
|
||||
}
|
||||
if s.bangumi != nil && s.bangumi.Enabled() {
|
||||
if m, err := s.bangumi.Search(ctx, query); err == nil && m != nil && metadataMatchCompatibleWithType(kind, m) {
|
||||
if m, err := s.bangumi.Search(ctx, query); err == nil && m != nil && metadataMatchCompatibleWithTheatrical(kind, isTheatrical, m) {
|
||||
m.Provider = "bangumi"
|
||||
return m
|
||||
} else if err != nil {
|
||||
|
||||
@@ -15,8 +15,41 @@ var (
|
||||
episodeTitleQueryRE = regexp.MustCompile(`^\s*第\s*[0-9一二三四五六七八九十百零两]+\s*[集期话話](?:\s*[上下])?\s*[::].+`)
|
||||
genericEpisodeWordsRE = regexp.MustCompile(`^\s*第\s*[集期话話]\s*$`)
|
||||
episodeReleaseTitleTagRE = regexp.MustCompile(`(?i)(?:^|[\s._-])s\d{1,2}e\d{1,3}(?:[\s._-]|$)`)
|
||||
patTheatricalTitle = regexp.MustCompile(`(?i)(?:剧场版|劇場版|动画电影|動畫電影|电影版|電影版|\bthe\s+movie\b|\bmovie\s*\d{1,2}\b)`)
|
||||
patTheatricalFolder = regexp.MustCompile(`(?i)[\\/](?:剧场版|劇場版|動畫電影|动画电影)[\\/]`)
|
||||
theatricalNoiseRE = regexp.MustCompile(`(?i)(?:剧场版|劇場版)\s*(?:第?\s*\d{1,3}\s*[部篇]?)?|电影版|電影版|动画电影|動畫電影`)
|
||||
)
|
||||
|
||||
func theatricalTitleVariants(raw string) []string {
|
||||
raw = strings.TrimSpace(raw)
|
||||
if raw == "" {
|
||||
return nil
|
||||
}
|
||||
if !theatricalNoiseRE.MatchString(raw) {
|
||||
return []string{raw}
|
||||
}
|
||||
stripped := theatricalNoiseRE.ReplaceAllString(raw, " ")
|
||||
stripped = strings.Join(strings.Fields(stripped), " ")
|
||||
if stripped != "" && !strings.EqualFold(stripped, raw) {
|
||||
return []string{stripped, raw}
|
||||
}
|
||||
return []string{raw}
|
||||
}
|
||||
|
||||
func mediaLooksLikeTheatricalFeature(m *model.Media) bool {
|
||||
if m == nil {
|
||||
return false
|
||||
}
|
||||
if m.SeasonNum > 0 || m.EpisodeNum > 0 {
|
||||
return false
|
||||
}
|
||||
if season, ep := ParseEpisode(m.Path); season > 0 || ep > 0 {
|
||||
return false
|
||||
}
|
||||
text := m.Title + " " + pathBaseSlash(m.Path)
|
||||
return patTheatricalTitle.MatchString(text) || patTheatricalFolder.MatchString(m.Path)
|
||||
}
|
||||
|
||||
func scrapeQueryCandidates(m *model.Media, lib *model.Library) []string {
|
||||
return scrapeQueryCandidatesWithNormalizer(m, lib, func(raw string) (string, int) {
|
||||
return CleanQuery(raw)
|
||||
@@ -33,31 +66,59 @@ func scrapeQueryCandidatesWithNormalizer(m *model.Media, lib *model.Library, cle
|
||||
seen := map[string]struct{}{}
|
||||
var out []string
|
||||
add := func(raw string) {
|
||||
cleaned, _ := clean(raw)
|
||||
if cleaned == "" {
|
||||
cleaned = strings.TrimSpace(raw)
|
||||
}
|
||||
for _, candidate := range titleCandidates(cleaned) {
|
||||
if unsafeAutomaticEpisodeQuery(candidate) {
|
||||
continue
|
||||
for _, variant := range theatricalTitleVariants(raw) {
|
||||
cleaned, _ := clean(variant)
|
||||
if cleaned == "" {
|
||||
cleaned = strings.TrimSpace(variant)
|
||||
}
|
||||
key := strings.ToLower(candidate)
|
||||
if _, ok := seen[key]; ok || candidate == "" {
|
||||
continue
|
||||
for _, candidate := range titleCandidates(cleaned) {
|
||||
if unsafeAutomaticEpisodeQuery(candidate) {
|
||||
continue
|
||||
}
|
||||
key := strings.ToLower(candidate)
|
||||
if _, ok := seen[key]; ok || candidate == "" {
|
||||
continue
|
||||
}
|
||||
seen[key] = struct{}{}
|
||||
out = append(out, candidate)
|
||||
}
|
||||
seen[key] = struct{}{}
|
||||
out = append(out, candidate)
|
||||
}
|
||||
}
|
||||
episodic := mediaIsEpisodic(m, lib)
|
||||
if lib != nil && episodic {
|
||||
add(seriesFolderTitle(m.Path, lib.Path))
|
||||
|
||||
isTheatrical := mediaLooksLikeTheatricalFeature(m)
|
||||
if isTheatrical {
|
||||
add(m.Title)
|
||||
add(m.Path)
|
||||
if lib != nil {
|
||||
seriesTitle := seriesFolderTitle(m.Path, lib.Path)
|
||||
if seriesTitle != "" {
|
||||
baseName := pathBaseSlash(m.Path)
|
||||
stem := mediaFileStem(baseName)
|
||||
if stem == "" {
|
||||
stem = strings.TrimSuffix(baseName, filepath.Ext(baseName))
|
||||
}
|
||||
cleanStem, _ := clean(stem)
|
||||
if cleanStem == "" {
|
||||
cleanStem = stem
|
||||
}
|
||||
if !strings.Contains(strings.ToLower(cleanStem), strings.ToLower(seriesTitle)) {
|
||||
add(seriesTitle + " " + cleanStem)
|
||||
}
|
||||
}
|
||||
add(mediaFolderTitle(m.Path, lib.Path))
|
||||
}
|
||||
} else {
|
||||
episodic := mediaIsEpisodic(m, lib)
|
||||
if lib != nil && episodic {
|
||||
add(seriesFolderTitle(m.Path, lib.Path))
|
||||
}
|
||||
if lib != nil {
|
||||
add(mediaFolderTitle(m.Path, lib.Path))
|
||||
}
|
||||
add(m.Title)
|
||||
add(m.Path)
|
||||
}
|
||||
if lib != nil {
|
||||
add(mediaFolderTitle(m.Path, lib.Path))
|
||||
}
|
||||
add(m.Title)
|
||||
add(m.Path)
|
||||
|
||||
if len(out) == 0 {
|
||||
base := pathBaseSlash(m.Path)
|
||||
out = append(out, strings.TrimSuffix(base, filepath.Ext(base)))
|
||||
@@ -106,6 +167,9 @@ func containsCJK(s string) bool {
|
||||
}
|
||||
|
||||
func mediaIsEpisodic(m *model.Media, lib *model.Library) bool {
|
||||
if m != nil && mediaLooksLikeTheatricalFeature(m) {
|
||||
return false
|
||||
}
|
||||
if m != nil && (m.SeasonNum > 0 || m.EpisodeNum > 0) {
|
||||
return true
|
||||
}
|
||||
|
||||
@@ -13,10 +13,10 @@ func mediaFolderTitle(mediaPath, libraryRoot string) string {
|
||||
if base == "" || base == "." {
|
||||
return ""
|
||||
}
|
||||
if isTechnicalMediaFolder(base) || strictSeasonFolderMatched(base) {
|
||||
dir = parentSlashPath(dir)
|
||||
continue
|
||||
}
|
||||
if isTechnicalMediaFolder(base) || strictSeasonFolderMatched(base) || isTheatricalFolder(base) {
|
||||
dir = parentSlashPath(dir)
|
||||
continue
|
||||
}
|
||||
if isMediaCollectionFolder(base) {
|
||||
dir = parentSlashPath(dir)
|
||||
continue
|
||||
@@ -104,7 +104,7 @@ func parentSlashPath(value string) string {
|
||||
|
||||
func seriesFolderTitle(mediaPath, libraryRoot string) string {
|
||||
dir := parentSlashPath(mediaPath)
|
||||
if strictSeasonFolderMatched(pathBaseSlash(dir)) {
|
||||
if strictSeasonFolderMatched(pathBaseSlash(dir)) || isTheatricalFolder(pathBaseSlash(dir)) {
|
||||
dir = parentSlashPath(dir)
|
||||
}
|
||||
if root := comparableLibraryRoot(libraryRoot); root != "" && sameSlashPath(dir, root) {
|
||||
@@ -114,12 +114,23 @@ func seriesFolderTitle(mediaPath, libraryRoot string) string {
|
||||
if base == "" || base == "." {
|
||||
return ""
|
||||
}
|
||||
if isGenericMediaCategoryFolder(base) || isTechnicalMediaFolder(base) || strictSeasonFolderMatched(base) {
|
||||
if isGenericMediaCategoryFolder(base) || isTechnicalMediaFolder(base) || strictSeasonFolderMatched(base) || isTheatricalFolder(base) {
|
||||
return ""
|
||||
}
|
||||
return base
|
||||
}
|
||||
|
||||
func isTheatricalFolder(name string) bool {
|
||||
key := strings.ToLower(strings.TrimSpace(name))
|
||||
key = strings.Trim(key, `\/`)
|
||||
switch key {
|
||||
case "剧场版", "劇場版", "动画电影", "動畫電影", "特别篇", "特別篇", "specials", "sp", "ova", "oad":
|
||||
return true
|
||||
default:
|
||||
return false
|
||||
}
|
||||
}
|
||||
|
||||
func libraryRootTitle(libraryRoot string) string {
|
||||
base := ""
|
||||
if info, ok := ParseCloudLibraryMount(libraryRoot); ok {
|
||||
|
||||
Reference in New Issue
Block a user