From b855e003458530018006b177a05f4b6b064bde22 Mon Sep 17 00:00:00 2001 From: truewhile <779943132@qq.com> Date: Thu, 10 Sep 2026 16:58:41 +0800 Subject: [PATCH] =?UTF-8?q?=E4=BC=98=E5=8C=96?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- .../service/organizer_directory_metadata.go | 2 +- internal/service/recognition_words.go | 79 ++++++++- .../service/recognition_words_cache_test.go | 49 ++++++ internal/service/recognition_words_rules.go | 80 ++++++--- internal/service/scraper.go | 7 +- internal/service/scraper_library.go | 29 ++- internal/service/scraper_lookup_cache.go | 126 +++++++++++++ internal/service/scraper_lookup_cache_test.go | 165 ++++++++++++++++++ internal/service/scraper_query_clean.go | 48 ++++- .../scraper_query_clean_regression_test.go | 77 ++++++++ internal/service/scraper_service.go | 5 + 11 files changed, 632 insertions(+), 35 deletions(-) create mode 100644 internal/service/recognition_words_cache_test.go create mode 100644 internal/service/scraper_lookup_cache.go create mode 100644 internal/service/scraper_lookup_cache_test.go create mode 100644 internal/service/scraper_query_clean_regression_test.go diff --git a/internal/service/organizer_directory_metadata.go b/internal/service/organizer_directory_metadata.go index 5a33187..c8ef342 100644 --- a/internal/service/organizer_directory_metadata.go +++ b/internal/service/organizer_directory_metadata.go @@ -55,7 +55,7 @@ func (o *OrganizerService) lookupOrganizeMetadata(ctx context.Context, src, sour continue } } - match := o.scraper.lookup(ctx, lib, media, candidate, year) + match := o.scraper.lookup(ctx, lib, media, candidate, year, false) if match != nil && strings.TrimSpace(match.Title) != "" { if !organizeMetadataMatchTrusted(candidate, year, match) { if cache != nil { diff --git a/internal/service/recognition_words.go b/internal/service/recognition_words.go index 194a283..5308265 100644 --- a/internal/service/recognition_words.go +++ b/internal/service/recognition_words.go @@ -5,8 +5,10 @@ import ( "encoding/json" "fmt" "net/http" + "regexp" "strconv" "strings" + "sync" "time" "go.uber.org/zap" @@ -61,8 +63,9 @@ func NewRecognitionWordsService(log *zap.Logger, repo *repository.Container) *Re } func (s *RecognitionWordsService) Config(ctx context.Context) RecognitionWordsConfig { - cfg := recognitionWordsConfig(ctx, s.repo) - cfg.RuleCount = len(parseRecognitionWordRules(recognitionWordsCombinedText(cfg))) + cfg, rules := recognitionWordsRuntime(ctx, s.repo) + cfg = cloneRecognitionWordsConfig(cfg) + cfg.RuleCount = len(rules) return cfg } @@ -80,7 +83,11 @@ func (s *RecognitionWordsService) SaveConfig(ctx context.Context, cfg Recognitio if err != nil { return err } - return s.repo.Setting.Set(ctx, RecognitionWordsSharedURLsKey, string(rawURLs)) + if err := s.repo.Setting.Set(ctx, RecognitionWordsSharedURLsKey, string(rawURLs)); err != nil { + return err + } + invalidateRecognitionWordsRuntime(s.repo) + return nil } func (s *RecognitionWordsService) SyncShared(ctx context.Context) (RecognitionWordsConfig, error) { @@ -104,6 +111,7 @@ func (s *RecognitionWordsService) SyncShared(ctx context.Context) (RecognitionWo if err := s.repo.Setting.Set(ctx, RecognitionWordsSyncedAtKey, now); err != nil { return cfg, err } + invalidateRecognitionWordsRuntime(s.repo) return s.Config(ctx), nil } @@ -120,11 +128,10 @@ func (s *RecognitionWordsService) Test(ctx context.Context, input string) Recogn } func ApplyRecognitionWords(ctx context.Context, repo *repository.Container, raw string) string { - cfg := recognitionWordsConfig(ctx, repo) - if !cfg.Enabled { + cfg, rules := recognitionWordsRuntime(ctx, repo) + if !cfg.Enabled || len(rules) == 0 { return raw } - rules := parseRecognitionWordRules(recognitionWordsCombinedText(cfg)) return applyRecognitionWordRules(raw, rules) } @@ -189,9 +196,69 @@ func normalizeRecognitionWordURLs(values []string) []string { type recognitionWordRule struct { raw string block string + blockRE *regexp.Regexp replaceFrom string replaceTo string + replaceRE *regexp.Regexp offsetLeft string offsetRight string offsetExpr string + offsetRE *regexp.Regexp +} + +// recognitionWordsRuntime caches the parsed rule set (with pre-compiled +// regexes) per repository. Reading five settings rows and parsing/recompiling +// the whole shared word list used to happen on every query candidate, which is +// the scraper's hottest path. Writes through SaveConfig/SyncShared invalidate +// the entry immediately; the TTL only bounds staleness for out-of-process edits. +type recognitionWordsRuntimeEntry struct { + repo *repository.Container + cfg RecognitionWordsConfig + rules []recognitionWordRule + expiresAt time.Time +} + +const recognitionWordsRuntimeTTL = 30 * time.Second + +var recognitionWordsRuntimeCache struct { + sync.RWMutex + entry recognitionWordsRuntimeEntry + valid bool +} + +func recognitionWordsRuntime(ctx context.Context, repo *repository.Container) (RecognitionWordsConfig, []recognitionWordRule) { + now := time.Now() + recognitionWordsRuntimeCache.RLock() + cached := recognitionWordsRuntimeCache.entry + hit := recognitionWordsRuntimeCache.valid && cached.repo == repo && now.Before(cached.expiresAt) + recognitionWordsRuntimeCache.RUnlock() + if hit { + return cached.cfg, cached.rules + } + + cfg := recognitionWordsConfig(ctx, repo) + rules := parseRecognitionWordRules(recognitionWordsCombinedText(cfg)) + recognitionWordsRuntimeCache.Lock() + recognitionWordsRuntimeCache.entry = recognitionWordsRuntimeEntry{ + repo: repo, + cfg: cfg, + rules: rules, + expiresAt: now.Add(recognitionWordsRuntimeTTL), + } + recognitionWordsRuntimeCache.valid = true + recognitionWordsRuntimeCache.Unlock() + return cfg, rules +} + +func invalidateRecognitionWordsRuntime(repo *repository.Container) { + recognitionWordsRuntimeCache.Lock() + if recognitionWordsRuntimeCache.entry.repo == repo { + recognitionWordsRuntimeCache.valid = false + } + recognitionWordsRuntimeCache.Unlock() +} + +func cloneRecognitionWordsConfig(cfg RecognitionWordsConfig) RecognitionWordsConfig { + cfg.SharedURLs = append([]string(nil), cfg.SharedURLs...) + return cfg } diff --git a/internal/service/recognition_words_cache_test.go b/internal/service/recognition_words_cache_test.go new file mode 100644 index 0000000..65d4149 --- /dev/null +++ b/internal/service/recognition_words_cache_test.go @@ -0,0 +1,49 @@ +package service + +import ( + "testing" + + "go.uber.org/zap" +) + +// The rule set is cached process-wide, so a config write must invalidate it +// immediately; otherwise the admin would have to wait out the TTL before a +// saved word list takes effect. +func TestRecognitionWordsCacheInvalidatedBySaveConfig(t *testing.T) { + repos := newOrganizerTestRepo(t) + svc := NewRecognitionWordsService(zap.NewNop(), repos) + + if err := svc.SaveConfig(t.Context(), RecognitionWordsConfig{ + Enabled: true, + LocalText: "BADWORD => 好标题", + }); err != nil { + t.Fatal(err) + } + if got := ApplyRecognitionWords(t.Context(), repos, "BADWORD"); got != "好标题" { + t.Fatalf("first apply = %q, want 好标题", got) + } + + if err := svc.SaveConfig(t.Context(), RecognitionWordsConfig{ + Enabled: true, + LocalText: "BADWORD => 新标题", + }); err != nil { + t.Fatal(err) + } + if got := ApplyRecognitionWords(t.Context(), repos, "BADWORD"); got != "新标题" { + t.Fatalf("cached rules were not invalidated after SaveConfig: got %q, want 新标题", got) + } +} + +func TestRecognitionWordsDisabledReturnsRawInput(t *testing.T) { + repos := newOrganizerTestRepo(t) + svc := NewRecognitionWordsService(zap.NewNop(), repos) + if err := svc.SaveConfig(t.Context(), RecognitionWordsConfig{ + Enabled: false, + LocalText: "BADWORD => 好标题", + }); err != nil { + t.Fatal(err) + } + if got := ApplyRecognitionWords(t.Context(), repos, "BADWORD"); got != "BADWORD" { + t.Fatalf("disabled recognition words changed input: got %q", got) + } +} diff --git a/internal/service/recognition_words_rules.go b/internal/service/recognition_words_rules.go index d8a95cc..c3602f1 100644 --- a/internal/service/recognition_words_rules.go +++ b/internal/service/recognition_words_rules.go @@ -31,6 +31,11 @@ func parseRecognitionWordRule(line string) recognitionWordRule { case strings.Contains(part, "<>") && strings.Contains(part, ">>"): beforeAfter := strings.SplitN(part, ">>", 2) bounds := strings.SplitN(beforeAfter[0], "<>", 2) + // A malformed rule without "<>" would otherwise index past the + // slice; word lists are fetched from the network, so stay defensive. + if len(bounds) < 2 { + continue + } rule.offsetLeft = strings.TrimSpace(bounds[0]) rule.offsetRight = strings.TrimSpace(bounds[1]) rule.offsetExpr = strings.TrimSpace(beforeAfter[1]) @@ -38,56 +43,89 @@ func parseRecognitionWordRule(line string) recognitionWordRule { rule.block = part } } + compileRecognitionWordRule(&rule) return rule } +var recognitionReplacementRE = regexp.MustCompile(`\\([0-9]+)`) + func normalizeRecognitionReplacement(value string) string { - re := regexp.MustCompile(`\\([0-9]+)`) - return re.ReplaceAllString(value, "$$$1") + return recognitionReplacementRE.ReplaceAllString(value, "$$$1") +} + +// compileRecognitionWordRule pre-compiles every pattern in a rule so the hot +// clean-query path never recompiles regexes per candidate. +func compileRecognitionWordRule(rule *recognitionWordRule) { + if rule == nil { + return + } + if rule.block != "" { + if re, err := regexp.Compile(rule.block); err == nil { + rule.blockRE = re + } + } + if rule.replaceFrom != "" { + if re, err := regexp.Compile(rule.replaceFrom); err == nil { + rule.replaceRE = re + } + } + if rule.offsetExpr != "" && (rule.offsetLeft != "" || rule.offsetRight != "") { + if re, err := compileRecognitionOffsetRE(rule.offsetLeft, rule.offsetRight); err == nil { + rule.offsetRE = re + } + } +} + +func compileRecognitionOffsetRE(left, right string) (*regexp.Regexp, error) { + leftPattern := firstNonEmpty(left, `^`) + rightPattern := firstNonEmpty(right, `$`) + return regexp.Compile(`(?i)(` + leftPattern + `)(\d{1,5})(` + rightPattern + `)`) } func applyRecognitionWordRules(raw string, rules []recognitionWordRule) string { out := strings.TrimSpace(raw) for _, rule := range rules { if rule.block != "" { - out = applyRecognitionBlock(out, rule.block) + out = applyRecognitionBlock(out, rule) } if rule.replaceFrom != "" { - out = applyRecognitionReplace(out, rule.replaceFrom, rule.replaceTo) + out = applyRecognitionReplace(out, rule) } if rule.offsetLeft != "" || rule.offsetRight != "" { - out = applyRecognitionOffset(out, rule.offsetLeft, rule.offsetRight, rule.offsetExpr) + out = applyRecognitionOffset(out, rule) } } return strings.Join(strings.Fields(out), " ") } -func applyRecognitionBlock(raw, block string) string { - if re, err := regexp.Compile(block); err == nil { - return re.ReplaceAllString(raw, " ") +func applyRecognitionBlock(raw string, rule recognitionWordRule) string { + if rule.blockRE != nil { + return rule.blockRE.ReplaceAllString(raw, " ") } - return strings.ReplaceAll(raw, block, " ") + return strings.ReplaceAll(raw, rule.block, " ") } -func applyRecognitionReplace(raw, from, to string) string { - if re, err := regexp.Compile(from); err == nil { - return re.ReplaceAllString(raw, to) +func applyRecognitionReplace(raw string, rule recognitionWordRule) string { + if rule.replaceRE != nil { + return rule.replaceRE.ReplaceAllString(raw, rule.replaceTo) } - return strings.ReplaceAll(raw, from, to) + return strings.ReplaceAll(raw, rule.replaceFrom, rule.replaceTo) } -func applyRecognitionOffset(raw, left, right, expr string) string { - if strings.TrimSpace(expr) == "" { +func applyRecognitionOffset(raw string, rule recognitionWordRule) string { + if strings.TrimSpace(rule.offsetExpr) == "" { return raw } - leftPattern := firstNonEmpty(left, `^`) - rightPattern := firstNonEmpty(right, `$`) - re, err := regexp.Compile(`(?i)(` + leftPattern + `)(\d{1,5})(` + rightPattern + `)`) - if err != nil { - return raw + re := rule.offsetRE + if re == nil { + compiled, err := compileRecognitionOffsetRE(rule.offsetLeft, rule.offsetRight) + if err != nil { + return raw + } + re = compiled } return re.ReplaceAllStringFunc(raw, func(match string) string { - return applyRecognitionOffsetMatch(re, match, expr) + return applyRecognitionOffsetMatch(re, match, rule.offsetExpr) }) } diff --git a/internal/service/scraper.go b/internal/service/scraper.go index d608c91..ad311e7 100644 --- a/internal/service/scraper.go +++ b/internal/service/scraper.go @@ -63,12 +63,17 @@ func (s *ScraperService) EnrichOneWithOptions(ctx context.Context, m *model.Medi } } + // Same stale-negative problem as the library path: a single-item retry + // (queue task / manual rescrape) must get a fresh provider round-trip. + if options.RetryNoMatch && s != nil { + s.lookupCache.clearNegatives() + } candidates := scrapeQueryCandidatesWithRecognition(ctx, s.repo, m, lib) var query string match := (*Match)(nil) for _, candidate := range candidates { query = candidate - candidateMatch := s.lookup(ctx, lib, m, candidate, year) + candidateMatch := s.lookup(ctx, lib, m, candidate, year, options.ForceRematch) if candidateMatch == nil { continue } diff --git a/internal/service/scraper_library.go b/internal/service/scraper_library.go index 331775a..62b7a24 100644 --- a/internal/service/scraper_library.go +++ b/internal/service/scraper_library.go @@ -13,7 +13,12 @@ import ( // lookup runs the provider chain after local NFO has been considered: // TMDb -> Douban -> Bangumi -> TheTVDB. Douban and Bangumi do not require API // keys; providers that are unavailable or return an error are skipped. -func (s *ScraperService) lookup(ctx context.Context, lib *model.Library, media *model.Media, query string, year int) *Match { +// +// Results are cached per (kind, theatrical, query, year) because every +// episode of a show produces the same candidate; bypassCache forces a fresh +// provider round-trip for user-triggered rematches but still refreshes the +// cache so later episodes reuse the corrected result. +func (s *ScraperService) lookup(ctx context.Context, lib *model.Library, media *model.Media, query string, year int, bypassCache bool) *Match { kind := "" if lib != nil { kind = lib.Type @@ -28,6 +33,21 @@ func (s *ScraperService) lookup(ctx context.Context, lib *model.Library, media * } else if (normalizeOrganizeMediaType(kind) != "movie" || explicitEpisode) && mediaIsEpisodic(media, lib) { kind = "tv" } + + cacheKey := scrapeLookupCacheKey(kind, query, year, isTheatrical) + if !bypassCache { + if cached, ok := s.lookupCache.get(cacheKey); ok { + return cached + } + } + match := s.lookupUncached(ctx, kind, isTheatrical, query, year) + // Always refresh the cache, even on bypassed (forced) lookups, so a + // corrected rematch replaces the stale entry for later episodes. + s.lookupCache.set(cacheKey, match) + return match +} + +func (s *ScraperService) lookupUncached(ctx context.Context, kind string, isTheatrical bool, query string, year int) *Match { if s.tmdb != nil && s.tmdb.Enabled() { if match := s.lookupAutomaticTMDb(ctx, kind, query, year); match != nil { match.Provider = "tmdb" @@ -107,6 +127,13 @@ func (s *ScraperService) EnrichLibraryDetailed(ctx context.Context, libraryID st func (s *ScraperService) EnrichLibraryDetailedWithOptions(ctx context.Context, libraryID string, options ScrapeOptions) (EnrichLibraryResult, error) { result := EnrichLibraryResult{LibraryID: libraryID} + // A manual retry must not reuse stale negative cache entries from the + // previous run, otherwise "重新刮削" looks like it did nothing. Positives + // are kept so the rest of the season still dedupes; the first re-queried + // episode refreshes the entry for the rows behind it. + if options.RetryNoMatch && s != nil { + s.lookupCache.clearNegatives() + } rows, err := s.scrapeCandidateRows(ctx, libraryID, options) if err != nil { return result, err diff --git a/internal/service/scraper_lookup_cache.go b/internal/service/scraper_lookup_cache.go new file mode 100644 index 0000000..730a669 --- /dev/null +++ b/internal/service/scraper_lookup_cache.go @@ -0,0 +1,126 @@ +package service + +import ( + "strconv" + "strings" + "sync" + "time" +) + +// A TV library scrapes one media row per episode, and every row in the same +// show computes the same query candidate (the series folder title). Without a +// cache each episode re-issues the identical TMDb/Douban/Bangumi/TheTVDB +// search, so a 100-episode season costs 100x the network calls it needs. +// +// The cache lives on the ScraperService instance (not a package global) so that +// two services with different provider configuration never share results. TTL +// bounds staleness for out-of-band edits; explicit invalidation is unnecessary +// because the key includes the effective media kind, theatrical flag, query +// and year. Negative entries use a shorter TTL so a failed query is retried +// sooner without manual intervention. +const ( + scrapeLookupCacheTTL = 10 * time.Minute + scrapeLookupNegativeCacheTTL = 2 * time.Minute + scrapeLookupCacheMaxItems = 1024 +) + +type scrapeLookupCache struct { + mu sync.Mutex + entries map[string]scrapeLookupCacheEntry +} + +type scrapeLookupCacheEntry struct { + match *Match + expiresAt time.Time +} + +func newScrapeLookupCache() *scrapeLookupCache { + return &scrapeLookupCache{entries: map[string]scrapeLookupCacheEntry{}} +} + +func scrapeLookupCacheKey(kind, query string, year int, isTheatrical bool) string { + theatrical := "0" + if isTheatrical { + theatrical = "1" + } + return strings.ToLower(strings.TrimSpace(kind)) + "|" + + theatrical + "|" + + strconv.Itoa(year) + "|" + + strings.ToLower(strings.TrimSpace(query)) +} + +// get reports whether the key is cached. A hit may carry a nil match, which +// means the provider chain already ran and found nothing (negative cache). +// Use clearNegative before a user-triggered retry so stale negative entries +// do not suppress the fresh provider round-trip. +func (c *scrapeLookupCache) get(key string) (*Match, bool) { + if c == nil || key == "" { + return nil, false + } + now := time.Now() + c.mu.Lock() + defer c.mu.Unlock() + item, ok := c.entries[key] + if !ok { + return nil, false + } + if now.After(item.expiresAt) { + delete(c.entries, key) + return nil, false + } + return cloneMatch(item.match), true +} + +func (c *scrapeLookupCache) set(key string, match *Match) { + if c == nil || key == "" { + return + } + now := time.Now() + c.mu.Lock() + defer c.mu.Unlock() + if len(c.entries) >= scrapeLookupCacheMaxItems { + for k, item := range c.entries { + if now.After(item.expiresAt) || len(c.entries) >= scrapeLookupCacheMaxItems { + delete(c.entries, k) + } + if len(c.entries) < scrapeLookupCacheMaxItems { + break + } + } + } + ttl := scrapeLookupCacheTTL + if match == nil { + ttl = scrapeLookupNegativeCacheTTL + } + c.entries[key] = scrapeLookupCacheEntry{match: cloneMatch(match), expiresAt: now.Add(ttl)} +} + +// clearNegatives drops cached misses so a user-triggered retry gets a fresh +// provider round-trip instead of reusing a stale negative entry. +func (c *scrapeLookupCache) clearNegatives() { + if c == nil { + return + } + c.mu.Lock() + defer c.mu.Unlock() + for k, item := range c.entries { + if item.match == nil { + delete(c.entries, k) + } + } +} + +// cloneMatch returns an independent copy so callers that mutate the result +// (localized title preference, local metadata merge, fanart artwork) cannot +// corrupt the cached entry or leak state between media rows. +func cloneMatch(m *Match) *Match { + if m == nil { + return nil + } + out := *m + out.Languages = append([]string(nil), m.Languages...) + out.Countries = append([]string(nil), m.Countries...) + out.Genres = append([]string(nil), m.Genres...) + out.Aliases = append([]string(nil), m.Aliases...) + return &out +} diff --git a/internal/service/scraper_lookup_cache_test.go b/internal/service/scraper_lookup_cache_test.go new file mode 100644 index 0000000..50ce96b --- /dev/null +++ b/internal/service/scraper_lookup_cache_test.go @@ -0,0 +1,165 @@ +package service + +import ( + "encoding/json" + "net/http" + "net/http/httptest" + "sync/atomic" + "testing" + + "github.com/glebarez/sqlite" + "go.uber.org/zap" + "gorm.io/gorm" + + "github.com/truewhile/MeBox/internal/config" + "github.com/truewhile/MeBox/internal/model" + "github.com/truewhile/MeBox/internal/repository" +) + +// Every episode of a show produces the same query candidate (the series folder +// title). Without the lookup cache each episode re-issues the identical search, +// so a whole season costs one request per episode. This asserts the provider is +// hit once and subsequent episodes reuse the cached match. +func TestEnrichLibraryReusesLookupCacheAcrossEpisodes(t *testing.T) { + var searchCalls int32 + upstream := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + w.Header().Set("Content-Type", "application/json") + switch r.URL.Path { + case "/search/tv": + if r.URL.Query().Get("query") != "折腰" { + _ = json.NewEncoder(w).Encode(map[string]any{"results": []any{}}) + return + } + atomic.AddInt32(&searchCalls, 1) + _ = json.NewEncoder(w).Encode(map[string]any{ + "results": []map[string]any{{ + "id": 296753, + "name": "折腰", + "overview": "正确的剧集条目", + "poster_path": "/zheyao.jpg", + "first_air_date": "2025-05-13", + "origin_country": []string{"CN"}, + }}, + }) + default: + http.NotFound(w, r) + } + })) + defer upstream.Close() + + db, err := gorm.Open(sqlite.Open("file::memory:?cache=shared"), &gorm.Config{}) + if err != nil { + t.Fatal(err) + } + if err := db.AutoMigrate(&model.Library{}, &model.Series{}, &model.Media{}); err != nil { + t.Fatal(err) + } + repos := repository.New(db) + cfg := &config.Config{} + cfg.Secrets.TMDbAPIKey = "test-key" + cfg.Secrets.TMDbAPIProxy = upstream.URL + cfg.Secrets.TMDbImageProxy = upstream.URL + "/images" + log := zap.NewNop() + scraper := NewScraperService(cfg, log, repos, NewTMDbProvider(cfg, log, nil), nil, nil, nil, NewHub(log)) + + lib := model.Library{Name: "OpenList · 刮削缓存测试库", Path: "cloud://openlist/scrape-lookup-cache", Type: "tv", Enabled: true} + if err := repos.DB.Create(&lib).Error; err != nil { + t.Fatal(err) + } + const episodeCount = 4 + for episode := 1; episode <= episodeCount; episode++ { + media := model.Media{ + LibraryID: lib.ID, + Title: "折腰", + Path: "cloud://openlist/scrape-lookup-cache/折腰 (2025)/Season 1/折腰.S01E0" + string(rune('0'+episode)) + ".mkv", + SeasonNum: 1, + EpisodeNum: episode, + ScrapeStatus: "pending", + } + if err := repos.DB.Create(&media).Error; err != nil { + t.Fatal(err) + } + } + + result, err := scraper.EnrichLibraryDetailedWithOptions(t.Context(), lib.ID, ScrapeOptions{}) + if err != nil { + t.Fatal(err) + } + if result.Matched != episodeCount { + t.Fatalf("matched = %d, want %d", result.Matched, episodeCount) + } + if calls := atomic.LoadInt32(&searchCalls); calls != 1 { + t.Fatalf("tmdb /search/tv called %d times for %d episodes; want 1 (cache should dedupe identical queries)", calls, episodeCount) + } +} + +// A cached match must be handed out as an independent copy: mutating one row's +// result (localized title preference, local metadata merge) must not bleed into +// the next row. +func TestScrapeLookupCacheReturnsIndependentCopies(t *testing.T) { + cache := newScrapeLookupCache() + original := &Match{ + Title: "折腰", + TMDbID: 296753, + Genres: []string{"剧情"}, + Countries: []string{"CN"}, + Aliases: []string{"Zhe Yao"}, + } + key := scrapeLookupCacheKey("tv", "折腰", 2025, false) + cache.set(key, original) + + first, ok := cache.get(key) + if !ok || first == nil { + t.Fatal("expected a cache hit") + } + first.Title = "mutated" + first.Genres[0] = "mutated" + first.Aliases = append(first.Aliases, "extra") + + second, ok := cache.get(key) + if !ok || second == nil { + t.Fatal("expected a second cache hit") + } + if second.Title != "折腰" || second.Genres[0] != "剧情" || len(second.Aliases) != 1 { + t.Fatalf("cached match was mutated through a returned copy: %+v", second) + } +} + +// Negative results are cached too: a query that matched nothing must not be +// re-issued for every remaining episode of the same show. +func TestScrapeLookupCacheStoresNegativeResults(t *testing.T) { + cache := newScrapeLookupCache() + key := scrapeLookupCacheKey("tv", "no-such-show", 0, false) + cache.set(key, nil) + if _, ok := cache.get(key); !ok { + t.Fatal("negative result should be cached to avoid repeated provider calls") + } +} + +// The theatrical flag is part of the key: a theatrical feature gets a tv +// fallback lookup, so its result must not be shared with (or returned for) +// the same query issued for a non-theatrical media row. +func TestScrapeLookupCacheKeySeparatesTheatrical(t *testing.T) { + plain := scrapeLookupCacheKey("movie", "query", 2024, false) + theatrical := scrapeLookupCacheKey("movie", "query", 2024, true) + if plain == theatrical { + t.Fatalf("theatrical flag must be part of the cache key: %q", plain) + } +} + +// A manual "retry no match" run clears stale negative entries so the provider +// chain is actually re-queried instead of short-circuiting on the cached miss. +func TestScrapeLookupCacheClearNegativesKeepsPositives(t *testing.T) { + cache := newScrapeLookupCache() + negKey := scrapeLookupCacheKey("tv", "missing", 0, false) + posKey := scrapeLookupCacheKey("tv", "found", 0, false) + cache.set(negKey, nil) + cache.set(posKey, &Match{Title: "found"}) + cache.clearNegatives() + if _, ok := cache.get(negKey); ok { + t.Fatal("negative entry should be cleared before a manual retry") + } + if got, ok := cache.get(posKey); !ok || got == nil || got.Title != "found" { + t.Fatalf("positive entry should survive clearNegatives: %+v", got) + } +} diff --git a/internal/service/scraper_query_clean.go b/internal/service/scraper_query_clean.go index 9888f99..bc208ad 100644 --- a/internal/service/scraper_query_clean.go +++ b/internal/service/scraper_query_clean.go @@ -24,11 +24,15 @@ var noiseTokens = []string{ "hkfree", "yify", "rarbg", "ettv", "fgt", "tgx", "ctrlhd", "ntb", "flux", "qhstudio", // 流媒体平台 / 字幕组 / 国家版本(动漫常见) - "netflix", "nf", "amzn", "hulu", "disney", "max", "hbo", + // 注意:不要把同时是常见英文单词的标记放进来(如 max / web / judas), + // 否则 "Mad Max"、"Web Therapy" 这类正常标题会被误删。裸 "WEB" 发布标记 + // 通常位于分辨率之后,由 releaseBoundary 截断规则处理;"WEB-DL" 则由 + // multiWordNoise 单独匹配。 + "netflix", "nf", "amzn", "hulu", "disney", "hbo", "linetv", "ourtv", "iqiyi", "youku", "bilibili", "qiyi", "krj", "atvp", "appletv", "apple-tv", "tx", "txweb", "crunchyroll", "funimation", "anidb", "horriblesubs", "subsplease", - "erai-raws", "judas", "asw", "smcat", "leopard-raws", "ohys-raws", "colortv", + "erai-raws", "asw", "smcat", "leopard-raws", "ohys-raws", "colortv", "mweb", "ubweb", "hhweb", "adweb", "chdweb", "kurosawa", "qhstudio", // 中文字幕标记 @@ -55,6 +59,20 @@ var releaseBoundaryTokenSet = map[string]struct{}{ "x264": {}, "x265": {}, "h264": {}, "h265": {}, "h266": {}, "hevc": {}, "avc": {}, "av1": {}, "vvc": {}, } +// weakReleaseBoundaryTokenSet are release tags that double as plausible title +// words. Before any release signal has been seen they are kept as part of the +// title ("Mad Max", "Web Therapy"), so a title is never truncated to nothing; +// after a real signal they behave like any other tag. +var weakReleaseBoundaryTokenSet = map[string]struct{}{ + "bd": {}, "dvd": {}, "web": {}, +} + +// releaseSignalToken marks the position of an extracted year. The year itself +// is not a title token, but its presence still proves the following tokens are +// release tags ("复仇者联盟4.2019.BD.1080p" must drop "BD"). A control rune is +// used so it can never collide with a real filename token. +const releaseSignalToken = "\u0001" + var dynamicReleaseBoundaryTokenRE = regexp.MustCompile(`(?i)^(?:\d{3,4}p|\d{2,3}fps)$`) // bracketedTag matches "[anything]", "(anything)" or "{anything}" segments. @@ -86,7 +104,9 @@ func CleanQuery(raw string) (title string, year int) { if m := yearPattern.FindStringSubmatch(lower); len(m) >= 2 { if v, err := strconv.Atoi(m[1]); err == nil { year = v - lower = strings.ReplaceAll(lower, m[1], " ") + // Keep a positional marker: the year is not a title token, but its + // presence arms release-tag truncation for what follows. + lower = strings.ReplaceAll(lower, m[1], " "+releaseSignalToken+" ") } } @@ -110,17 +130,35 @@ func CleanQuery(raw string) (title string, year int) { } // 拆分后丢掉过短(≤1)且全为 ASCII 数字 / 字母的"碎片",避免 // 「2」「0」「v」之类残留干扰 TMDb 搜索。中文字符不算碎片。 + // + // seenReleaseBoundary 只有在遇到分辨率/编码等强标记后才置位,置位后其余 + // ASCII 词一律视为发布尾巴丢弃;releaseSignalled 则宽松得多,年份标记也会 + // 置位,它只用来武装弱标记(bd/dvd/web)。这样 "Web Therapy" 不会被截空, + // 而 "2019.Avatar.1080p" 这种年份前置的标题也不会因为年份标记把后面的 + // 真实标题词当成尾巴丢掉。 out := make([]string, 0, 8) seenReleaseBoundary := false + releaseSignalled := false for _, w := range strings.Fields(lower) { + if w == releaseSignalToken { + releaseSignalled = true + continue + } if dynamicReleaseBoundaryTokenRE.MatchString(w) { + releaseSignalled = true seenReleaseBoundary = true continue } - if _, ok := noiseTokenSet[w]; ok { - if _, boundary := releaseBoundaryTokenSet[w]; boundary { + if _, boundary := releaseBoundaryTokenSet[w]; boundary { + if _, weak := weakReleaseBoundaryTokenSet[w]; weak && !releaseSignalled { + // Ambiguous tag that is also a plausible title word; keep it + // until a real release signal proves we are past the title. + } else { + releaseSignalled = true seenReleaseBoundary = true + continue } + } else if _, ok := noiseTokenSet[w]; ok { continue } if seenReleaseBoundary && isASCIIWord(w) { diff --git a/internal/service/scraper_query_clean_regression_test.go b/internal/service/scraper_query_clean_regression_test.go new file mode 100644 index 0000000..8939757 --- /dev/null +++ b/internal/service/scraper_query_clean_regression_test.go @@ -0,0 +1,77 @@ +package service + +import "testing" + +// CleanQuery must not delete words that are also ordinary English title words +// just because a release group reused them as a tag. Regression guard: "max" +// and "web" used to be unconditional noise/boundary tokens, which turned +// "Mad Max" into "mad" and "Web Therapy" into an empty query. +func TestCleanQueryKeepsCommonEnglishTitleWords(t *testing.T) { + cases := []struct { + in string + wantTitle string + wantYear int + }{ + {"Mad.Max.1979.1080p.BluRay.mkv", "mad max", 1979}, + {"Max.Payne.2008.1080p.WEB-DL.mkv", "max payne", 2008}, + {"Web.Therapy.S01E01.1080p.WEB-DL.mkv", "web therapy", 0}, + {"The.Web.2019.1080p.mkv", "the web", 2019}, + } + for _, tc := range cases { + t.Run(tc.in, func(t *testing.T) { + gotTitle, gotYear := CleanQuery(tc.in) + if gotTitle != tc.wantTitle || gotYear != tc.wantYear { + t.Errorf("CleanQuery(%q) = (%q, %d), want (%q, %d)", + tc.in, gotTitle, gotYear, tc.wantTitle, tc.wantYear) + } + }) + } +} + +// A title consisting only of an ambiguous tag plus a release tail must not be +// truncated to nothing; the tag stays so the query still has a chance. +func TestCleanQueryDoesNotTruncateAmbiguousTagToNothing(t *testing.T) { + for _, in := range []string{"Web.1080p.WEB-DL.mkv", "BD.720p.mkv"} { + title, _ := CleanQuery(in) + if title == "" { + t.Errorf("CleanQuery(%q) returned an empty title", in) + } + } +} + +// A year-prefixed filename must keep its title: the year arms weak-tag +// truncation but must not discard the real title words that follow it. +func TestCleanQueryKeepsTitleAfterYearPrefix(t *testing.T) { + cases := map[string]string{ + "2019.Avatar.1080p.BluRay.mkv": "avatar", + "2024.Dune.Part.Two.2160p.WEB.mkv": "dune part two", + } + for in, want := range cases { + t.Run(in, func(t *testing.T) { + got, year := CleanQuery(in) + if got != want { + t.Errorf("CleanQuery(%q) = (%q, %d), want title %q", in, got, year, want) + } + }) + } +} + +// Release-tag truncation must still fire once a real release signal (an +// extracted year, resolution or codec) has been seen, so tags after it are +// dropped as before. +func TestCleanQueryStillDropsTagsAfterReleaseSignal(t *testing.T) { + cases := map[string]string{ + "复仇者联盟4.2019.BD.1080p.mkv": "复仇者联盟4", + "The.Matrix.1999.1080p.WEB-DL.H265.mp4": "the matrix", + "Oppenheimer.2023.2160p.UHD.BluRay.mkv": "oppenheimer", + "Interstellar.2014.4k.hdr.dts.atmos.mkv": "interstellar", + } + for in, want := range cases { + t.Run(in, func(t *testing.T) { + got, _ := CleanQuery(in) + if got != want { + t.Errorf("CleanQuery(%q) = %q, want %q", in, got, want) + } + }) + } +} diff --git a/internal/service/scraper_service.go b/internal/service/scraper_service.go index 91b77a8..083718c 100644 --- a/internal/service/scraper_service.go +++ b/internal/service/scraper_service.go @@ -26,6 +26,10 @@ type ScraperService struct { cache *RuntimeCacheService images *ImageProxy + // Caches provider-chain results keyed by media kind/query/year. Per-instance + // so services with different provider config never share results. + lookupCache *scrapeLookupCache + // Serializes final sidecar replacement. Windows cannot rename over an // existing file, and concurrent scrapes can target the same sidecar. artworkWriteMu sync.Mutex @@ -50,6 +54,7 @@ func NewScraperService( return &ScraperService{ cfg: cfg, log: log, repo: repo, tmdb: tmdb, bangumi: bangumi, thetvdb: thetvdb, fanart: fanart, adult: adultProvider, hub: hub, + lookupCache: newScrapeLookupCache(), } }