mirror of
https://github.com/truewhile/MeBox.git
synced 2026-10-08 14:26:37 +08:00
优化
This commit is contained in:
@@ -55,7 +55,7 @@ func (o *OrganizerService) lookupOrganizeMetadata(ctx context.Context, src, sour
|
|||||||
continue
|
continue
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
match := o.scraper.lookup(ctx, lib, media, candidate, year)
|
match := o.scraper.lookup(ctx, lib, media, candidate, year, false)
|
||||||
if match != nil && strings.TrimSpace(match.Title) != "" {
|
if match != nil && strings.TrimSpace(match.Title) != "" {
|
||||||
if !organizeMetadataMatchTrusted(candidate, year, match) {
|
if !organizeMetadataMatchTrusted(candidate, year, match) {
|
||||||
if cache != nil {
|
if cache != nil {
|
||||||
|
|||||||
@@ -5,8 +5,10 @@ import (
|
|||||||
"encoding/json"
|
"encoding/json"
|
||||||
"fmt"
|
"fmt"
|
||||||
"net/http"
|
"net/http"
|
||||||
|
"regexp"
|
||||||
"strconv"
|
"strconv"
|
||||||
"strings"
|
"strings"
|
||||||
|
"sync"
|
||||||
"time"
|
"time"
|
||||||
|
|
||||||
"go.uber.org/zap"
|
"go.uber.org/zap"
|
||||||
@@ -61,8 +63,9 @@ func NewRecognitionWordsService(log *zap.Logger, repo *repository.Container) *Re
|
|||||||
}
|
}
|
||||||
|
|
||||||
func (s *RecognitionWordsService) Config(ctx context.Context) RecognitionWordsConfig {
|
func (s *RecognitionWordsService) Config(ctx context.Context) RecognitionWordsConfig {
|
||||||
cfg := recognitionWordsConfig(ctx, s.repo)
|
cfg, rules := recognitionWordsRuntime(ctx, s.repo)
|
||||||
cfg.RuleCount = len(parseRecognitionWordRules(recognitionWordsCombinedText(cfg)))
|
cfg = cloneRecognitionWordsConfig(cfg)
|
||||||
|
cfg.RuleCount = len(rules)
|
||||||
return cfg
|
return cfg
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -80,7 +83,11 @@ func (s *RecognitionWordsService) SaveConfig(ctx context.Context, cfg Recognitio
|
|||||||
if err != nil {
|
if err != nil {
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
return s.repo.Setting.Set(ctx, RecognitionWordsSharedURLsKey, string(rawURLs))
|
if err := s.repo.Setting.Set(ctx, RecognitionWordsSharedURLsKey, string(rawURLs)); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
invalidateRecognitionWordsRuntime(s.repo)
|
||||||
|
return nil
|
||||||
}
|
}
|
||||||
|
|
||||||
func (s *RecognitionWordsService) SyncShared(ctx context.Context) (RecognitionWordsConfig, error) {
|
func (s *RecognitionWordsService) SyncShared(ctx context.Context) (RecognitionWordsConfig, error) {
|
||||||
@@ -104,6 +111,7 @@ func (s *RecognitionWordsService) SyncShared(ctx context.Context) (RecognitionWo
|
|||||||
if err := s.repo.Setting.Set(ctx, RecognitionWordsSyncedAtKey, now); err != nil {
|
if err := s.repo.Setting.Set(ctx, RecognitionWordsSyncedAtKey, now); err != nil {
|
||||||
return cfg, err
|
return cfg, err
|
||||||
}
|
}
|
||||||
|
invalidateRecognitionWordsRuntime(s.repo)
|
||||||
return s.Config(ctx), nil
|
return s.Config(ctx), nil
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -120,11 +128,10 @@ func (s *RecognitionWordsService) Test(ctx context.Context, input string) Recogn
|
|||||||
}
|
}
|
||||||
|
|
||||||
func ApplyRecognitionWords(ctx context.Context, repo *repository.Container, raw string) string {
|
func ApplyRecognitionWords(ctx context.Context, repo *repository.Container, raw string) string {
|
||||||
cfg := recognitionWordsConfig(ctx, repo)
|
cfg, rules := recognitionWordsRuntime(ctx, repo)
|
||||||
if !cfg.Enabled {
|
if !cfg.Enabled || len(rules) == 0 {
|
||||||
return raw
|
return raw
|
||||||
}
|
}
|
||||||
rules := parseRecognitionWordRules(recognitionWordsCombinedText(cfg))
|
|
||||||
return applyRecognitionWordRules(raw, rules)
|
return applyRecognitionWordRules(raw, rules)
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -189,9 +196,69 @@ func normalizeRecognitionWordURLs(values []string) []string {
|
|||||||
type recognitionWordRule struct {
|
type recognitionWordRule struct {
|
||||||
raw string
|
raw string
|
||||||
block string
|
block string
|
||||||
|
blockRE *regexp.Regexp
|
||||||
replaceFrom string
|
replaceFrom string
|
||||||
replaceTo string
|
replaceTo string
|
||||||
|
replaceRE *regexp.Regexp
|
||||||
offsetLeft string
|
offsetLeft string
|
||||||
offsetRight string
|
offsetRight string
|
||||||
offsetExpr string
|
offsetExpr string
|
||||||
|
offsetRE *regexp.Regexp
|
||||||
|
}
|
||||||
|
|
||||||
|
// recognitionWordsRuntime caches the parsed rule set (with pre-compiled
|
||||||
|
// regexes) per repository. Reading five settings rows and parsing/recompiling
|
||||||
|
// the whole shared word list used to happen on every query candidate, which is
|
||||||
|
// the scraper's hottest path. Writes through SaveConfig/SyncShared invalidate
|
||||||
|
// the entry immediately; the TTL only bounds staleness for out-of-process edits.
|
||||||
|
type recognitionWordsRuntimeEntry struct {
|
||||||
|
repo *repository.Container
|
||||||
|
cfg RecognitionWordsConfig
|
||||||
|
rules []recognitionWordRule
|
||||||
|
expiresAt time.Time
|
||||||
|
}
|
||||||
|
|
||||||
|
const recognitionWordsRuntimeTTL = 30 * time.Second
|
||||||
|
|
||||||
|
var recognitionWordsRuntimeCache struct {
|
||||||
|
sync.RWMutex
|
||||||
|
entry recognitionWordsRuntimeEntry
|
||||||
|
valid bool
|
||||||
|
}
|
||||||
|
|
||||||
|
func recognitionWordsRuntime(ctx context.Context, repo *repository.Container) (RecognitionWordsConfig, []recognitionWordRule) {
|
||||||
|
now := time.Now()
|
||||||
|
recognitionWordsRuntimeCache.RLock()
|
||||||
|
cached := recognitionWordsRuntimeCache.entry
|
||||||
|
hit := recognitionWordsRuntimeCache.valid && cached.repo == repo && now.Before(cached.expiresAt)
|
||||||
|
recognitionWordsRuntimeCache.RUnlock()
|
||||||
|
if hit {
|
||||||
|
return cached.cfg, cached.rules
|
||||||
|
}
|
||||||
|
|
||||||
|
cfg := recognitionWordsConfig(ctx, repo)
|
||||||
|
rules := parseRecognitionWordRules(recognitionWordsCombinedText(cfg))
|
||||||
|
recognitionWordsRuntimeCache.Lock()
|
||||||
|
recognitionWordsRuntimeCache.entry = recognitionWordsRuntimeEntry{
|
||||||
|
repo: repo,
|
||||||
|
cfg: cfg,
|
||||||
|
rules: rules,
|
||||||
|
expiresAt: now.Add(recognitionWordsRuntimeTTL),
|
||||||
|
}
|
||||||
|
recognitionWordsRuntimeCache.valid = true
|
||||||
|
recognitionWordsRuntimeCache.Unlock()
|
||||||
|
return cfg, rules
|
||||||
|
}
|
||||||
|
|
||||||
|
func invalidateRecognitionWordsRuntime(repo *repository.Container) {
|
||||||
|
recognitionWordsRuntimeCache.Lock()
|
||||||
|
if recognitionWordsRuntimeCache.entry.repo == repo {
|
||||||
|
recognitionWordsRuntimeCache.valid = false
|
||||||
|
}
|
||||||
|
recognitionWordsRuntimeCache.Unlock()
|
||||||
|
}
|
||||||
|
|
||||||
|
func cloneRecognitionWordsConfig(cfg RecognitionWordsConfig) RecognitionWordsConfig {
|
||||||
|
cfg.SharedURLs = append([]string(nil), cfg.SharedURLs...)
|
||||||
|
return cfg
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -0,0 +1,49 @@
|
|||||||
|
package service
|
||||||
|
|
||||||
|
import (
|
||||||
|
"testing"
|
||||||
|
|
||||||
|
"go.uber.org/zap"
|
||||||
|
)
|
||||||
|
|
||||||
|
// The rule set is cached process-wide, so a config write must invalidate it
|
||||||
|
// immediately; otherwise the admin would have to wait out the TTL before a
|
||||||
|
// saved word list takes effect.
|
||||||
|
func TestRecognitionWordsCacheInvalidatedBySaveConfig(t *testing.T) {
|
||||||
|
repos := newOrganizerTestRepo(t)
|
||||||
|
svc := NewRecognitionWordsService(zap.NewNop(), repos)
|
||||||
|
|
||||||
|
if err := svc.SaveConfig(t.Context(), RecognitionWordsConfig{
|
||||||
|
Enabled: true,
|
||||||
|
LocalText: "BADWORD => 好标题",
|
||||||
|
}); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if got := ApplyRecognitionWords(t.Context(), repos, "BADWORD"); got != "好标题" {
|
||||||
|
t.Fatalf("first apply = %q, want 好标题", got)
|
||||||
|
}
|
||||||
|
|
||||||
|
if err := svc.SaveConfig(t.Context(), RecognitionWordsConfig{
|
||||||
|
Enabled: true,
|
||||||
|
LocalText: "BADWORD => 新标题",
|
||||||
|
}); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if got := ApplyRecognitionWords(t.Context(), repos, "BADWORD"); got != "新标题" {
|
||||||
|
t.Fatalf("cached rules were not invalidated after SaveConfig: got %q, want 新标题", got)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRecognitionWordsDisabledReturnsRawInput(t *testing.T) {
|
||||||
|
repos := newOrganizerTestRepo(t)
|
||||||
|
svc := NewRecognitionWordsService(zap.NewNop(), repos)
|
||||||
|
if err := svc.SaveConfig(t.Context(), RecognitionWordsConfig{
|
||||||
|
Enabled: false,
|
||||||
|
LocalText: "BADWORD => 好标题",
|
||||||
|
}); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if got := ApplyRecognitionWords(t.Context(), repos, "BADWORD"); got != "BADWORD" {
|
||||||
|
t.Fatalf("disabled recognition words changed input: got %q", got)
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -31,6 +31,11 @@ func parseRecognitionWordRule(line string) recognitionWordRule {
|
|||||||
case strings.Contains(part, "<>") && strings.Contains(part, ">>"):
|
case strings.Contains(part, "<>") && strings.Contains(part, ">>"):
|
||||||
beforeAfter := strings.SplitN(part, ">>", 2)
|
beforeAfter := strings.SplitN(part, ">>", 2)
|
||||||
bounds := strings.SplitN(beforeAfter[0], "<>", 2)
|
bounds := strings.SplitN(beforeAfter[0], "<>", 2)
|
||||||
|
// A malformed rule without "<>" would otherwise index past the
|
||||||
|
// slice; word lists are fetched from the network, so stay defensive.
|
||||||
|
if len(bounds) < 2 {
|
||||||
|
continue
|
||||||
|
}
|
||||||
rule.offsetLeft = strings.TrimSpace(bounds[0])
|
rule.offsetLeft = strings.TrimSpace(bounds[0])
|
||||||
rule.offsetRight = strings.TrimSpace(bounds[1])
|
rule.offsetRight = strings.TrimSpace(bounds[1])
|
||||||
rule.offsetExpr = strings.TrimSpace(beforeAfter[1])
|
rule.offsetExpr = strings.TrimSpace(beforeAfter[1])
|
||||||
@@ -38,56 +43,89 @@ func parseRecognitionWordRule(line string) recognitionWordRule {
|
|||||||
rule.block = part
|
rule.block = part
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
compileRecognitionWordRule(&rule)
|
||||||
return rule
|
return rule
|
||||||
}
|
}
|
||||||
|
|
||||||
|
var recognitionReplacementRE = regexp.MustCompile(`\\([0-9]+)`)
|
||||||
|
|
||||||
func normalizeRecognitionReplacement(value string) string {
|
func normalizeRecognitionReplacement(value string) string {
|
||||||
re := regexp.MustCompile(`\\([0-9]+)`)
|
return recognitionReplacementRE.ReplaceAllString(value, "$$$1")
|
||||||
return re.ReplaceAllString(value, "$$$1")
|
}
|
||||||
|
|
||||||
|
// compileRecognitionWordRule pre-compiles every pattern in a rule so the hot
|
||||||
|
// clean-query path never recompiles regexes per candidate.
|
||||||
|
func compileRecognitionWordRule(rule *recognitionWordRule) {
|
||||||
|
if rule == nil {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if rule.block != "" {
|
||||||
|
if re, err := regexp.Compile(rule.block); err == nil {
|
||||||
|
rule.blockRE = re
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if rule.replaceFrom != "" {
|
||||||
|
if re, err := regexp.Compile(rule.replaceFrom); err == nil {
|
||||||
|
rule.replaceRE = re
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if rule.offsetExpr != "" && (rule.offsetLeft != "" || rule.offsetRight != "") {
|
||||||
|
if re, err := compileRecognitionOffsetRE(rule.offsetLeft, rule.offsetRight); err == nil {
|
||||||
|
rule.offsetRE = re
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func compileRecognitionOffsetRE(left, right string) (*regexp.Regexp, error) {
|
||||||
|
leftPattern := firstNonEmpty(left, `^`)
|
||||||
|
rightPattern := firstNonEmpty(right, `$`)
|
||||||
|
return regexp.Compile(`(?i)(` + leftPattern + `)(\d{1,5})(` + rightPattern + `)`)
|
||||||
}
|
}
|
||||||
|
|
||||||
func applyRecognitionWordRules(raw string, rules []recognitionWordRule) string {
|
func applyRecognitionWordRules(raw string, rules []recognitionWordRule) string {
|
||||||
out := strings.TrimSpace(raw)
|
out := strings.TrimSpace(raw)
|
||||||
for _, rule := range rules {
|
for _, rule := range rules {
|
||||||
if rule.block != "" {
|
if rule.block != "" {
|
||||||
out = applyRecognitionBlock(out, rule.block)
|
out = applyRecognitionBlock(out, rule)
|
||||||
}
|
}
|
||||||
if rule.replaceFrom != "" {
|
if rule.replaceFrom != "" {
|
||||||
out = applyRecognitionReplace(out, rule.replaceFrom, rule.replaceTo)
|
out = applyRecognitionReplace(out, rule)
|
||||||
}
|
}
|
||||||
if rule.offsetLeft != "" || rule.offsetRight != "" {
|
if rule.offsetLeft != "" || rule.offsetRight != "" {
|
||||||
out = applyRecognitionOffset(out, rule.offsetLeft, rule.offsetRight, rule.offsetExpr)
|
out = applyRecognitionOffset(out, rule)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
return strings.Join(strings.Fields(out), " ")
|
return strings.Join(strings.Fields(out), " ")
|
||||||
}
|
}
|
||||||
|
|
||||||
func applyRecognitionBlock(raw, block string) string {
|
func applyRecognitionBlock(raw string, rule recognitionWordRule) string {
|
||||||
if re, err := regexp.Compile(block); err == nil {
|
if rule.blockRE != nil {
|
||||||
return re.ReplaceAllString(raw, " ")
|
return rule.blockRE.ReplaceAllString(raw, " ")
|
||||||
}
|
}
|
||||||
return strings.ReplaceAll(raw, block, " ")
|
return strings.ReplaceAll(raw, rule.block, " ")
|
||||||
}
|
}
|
||||||
|
|
||||||
func applyRecognitionReplace(raw, from, to string) string {
|
func applyRecognitionReplace(raw string, rule recognitionWordRule) string {
|
||||||
if re, err := regexp.Compile(from); err == nil {
|
if rule.replaceRE != nil {
|
||||||
return re.ReplaceAllString(raw, to)
|
return rule.replaceRE.ReplaceAllString(raw, rule.replaceTo)
|
||||||
}
|
}
|
||||||
return strings.ReplaceAll(raw, from, to)
|
return strings.ReplaceAll(raw, rule.replaceFrom, rule.replaceTo)
|
||||||
}
|
}
|
||||||
|
|
||||||
func applyRecognitionOffset(raw, left, right, expr string) string {
|
func applyRecognitionOffset(raw string, rule recognitionWordRule) string {
|
||||||
if strings.TrimSpace(expr) == "" {
|
if strings.TrimSpace(rule.offsetExpr) == "" {
|
||||||
return raw
|
return raw
|
||||||
}
|
}
|
||||||
leftPattern := firstNonEmpty(left, `^`)
|
re := rule.offsetRE
|
||||||
rightPattern := firstNonEmpty(right, `$`)
|
if re == nil {
|
||||||
re, err := regexp.Compile(`(?i)(` + leftPattern + `)(\d{1,5})(` + rightPattern + `)`)
|
compiled, err := compileRecognitionOffsetRE(rule.offsetLeft, rule.offsetRight)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return raw
|
return raw
|
||||||
|
}
|
||||||
|
re = compiled
|
||||||
}
|
}
|
||||||
return re.ReplaceAllStringFunc(raw, func(match string) string {
|
return re.ReplaceAllStringFunc(raw, func(match string) string {
|
||||||
return applyRecognitionOffsetMatch(re, match, expr)
|
return applyRecognitionOffsetMatch(re, match, rule.offsetExpr)
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -63,12 +63,17 @@ func (s *ScraperService) EnrichOneWithOptions(ctx context.Context, m *model.Medi
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Same stale-negative problem as the library path: a single-item retry
|
||||||
|
// (queue task / manual rescrape) must get a fresh provider round-trip.
|
||||||
|
if options.RetryNoMatch && s != nil {
|
||||||
|
s.lookupCache.clearNegatives()
|
||||||
|
}
|
||||||
candidates := scrapeQueryCandidatesWithRecognition(ctx, s.repo, m, lib)
|
candidates := scrapeQueryCandidatesWithRecognition(ctx, s.repo, m, lib)
|
||||||
var query string
|
var query string
|
||||||
match := (*Match)(nil)
|
match := (*Match)(nil)
|
||||||
for _, candidate := range candidates {
|
for _, candidate := range candidates {
|
||||||
query = candidate
|
query = candidate
|
||||||
candidateMatch := s.lookup(ctx, lib, m, candidate, year)
|
candidateMatch := s.lookup(ctx, lib, m, candidate, year, options.ForceRematch)
|
||||||
if candidateMatch == nil {
|
if candidateMatch == nil {
|
||||||
continue
|
continue
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -13,7 +13,12 @@ import (
|
|||||||
// lookup runs the provider chain after local NFO has been considered:
|
// lookup runs the provider chain after local NFO has been considered:
|
||||||
// TMDb -> Douban -> Bangumi -> TheTVDB. Douban and Bangumi do not require API
|
// TMDb -> Douban -> Bangumi -> TheTVDB. Douban and Bangumi do not require API
|
||||||
// keys; providers that are unavailable or return an error are skipped.
|
// keys; providers that are unavailable or return an error are skipped.
|
||||||
func (s *ScraperService) lookup(ctx context.Context, lib *model.Library, media *model.Media, query string, year int) *Match {
|
//
|
||||||
|
// Results are cached per (kind, theatrical, query, year) because every
|
||||||
|
// episode of a show produces the same candidate; bypassCache forces a fresh
|
||||||
|
// provider round-trip for user-triggered rematches but still refreshes the
|
||||||
|
// cache so later episodes reuse the corrected result.
|
||||||
|
func (s *ScraperService) lookup(ctx context.Context, lib *model.Library, media *model.Media, query string, year int, bypassCache bool) *Match {
|
||||||
kind := ""
|
kind := ""
|
||||||
if lib != nil {
|
if lib != nil {
|
||||||
kind = lib.Type
|
kind = lib.Type
|
||||||
@@ -28,6 +33,21 @@ func (s *ScraperService) lookup(ctx context.Context, lib *model.Library, media *
|
|||||||
} else if (normalizeOrganizeMediaType(kind) != "movie" || explicitEpisode) && mediaIsEpisodic(media, lib) {
|
} else if (normalizeOrganizeMediaType(kind) != "movie" || explicitEpisode) && mediaIsEpisodic(media, lib) {
|
||||||
kind = "tv"
|
kind = "tv"
|
||||||
}
|
}
|
||||||
|
|
||||||
|
cacheKey := scrapeLookupCacheKey(kind, query, year, isTheatrical)
|
||||||
|
if !bypassCache {
|
||||||
|
if cached, ok := s.lookupCache.get(cacheKey); ok {
|
||||||
|
return cached
|
||||||
|
}
|
||||||
|
}
|
||||||
|
match := s.lookupUncached(ctx, kind, isTheatrical, query, year)
|
||||||
|
// Always refresh the cache, even on bypassed (forced) lookups, so a
|
||||||
|
// corrected rematch replaces the stale entry for later episodes.
|
||||||
|
s.lookupCache.set(cacheKey, match)
|
||||||
|
return match
|
||||||
|
}
|
||||||
|
|
||||||
|
func (s *ScraperService) lookupUncached(ctx context.Context, kind string, isTheatrical bool, query string, year int) *Match {
|
||||||
if s.tmdb != nil && s.tmdb.Enabled() {
|
if s.tmdb != nil && s.tmdb.Enabled() {
|
||||||
if match := s.lookupAutomaticTMDb(ctx, kind, query, year); match != nil {
|
if match := s.lookupAutomaticTMDb(ctx, kind, query, year); match != nil {
|
||||||
match.Provider = "tmdb"
|
match.Provider = "tmdb"
|
||||||
@@ -107,6 +127,13 @@ func (s *ScraperService) EnrichLibraryDetailed(ctx context.Context, libraryID st
|
|||||||
|
|
||||||
func (s *ScraperService) EnrichLibraryDetailedWithOptions(ctx context.Context, libraryID string, options ScrapeOptions) (EnrichLibraryResult, error) {
|
func (s *ScraperService) EnrichLibraryDetailedWithOptions(ctx context.Context, libraryID string, options ScrapeOptions) (EnrichLibraryResult, error) {
|
||||||
result := EnrichLibraryResult{LibraryID: libraryID}
|
result := EnrichLibraryResult{LibraryID: libraryID}
|
||||||
|
// A manual retry must not reuse stale negative cache entries from the
|
||||||
|
// previous run, otherwise "重新刮削" looks like it did nothing. Positives
|
||||||
|
// are kept so the rest of the season still dedupes; the first re-queried
|
||||||
|
// episode refreshes the entry for the rows behind it.
|
||||||
|
if options.RetryNoMatch && s != nil {
|
||||||
|
s.lookupCache.clearNegatives()
|
||||||
|
}
|
||||||
rows, err := s.scrapeCandidateRows(ctx, libraryID, options)
|
rows, err := s.scrapeCandidateRows(ctx, libraryID, options)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return result, err
|
return result, err
|
||||||
|
|||||||
@@ -0,0 +1,126 @@
|
|||||||
|
package service
|
||||||
|
|
||||||
|
import (
|
||||||
|
"strconv"
|
||||||
|
"strings"
|
||||||
|
"sync"
|
||||||
|
"time"
|
||||||
|
)
|
||||||
|
|
||||||
|
// A TV library scrapes one media row per episode, and every row in the same
|
||||||
|
// show computes the same query candidate (the series folder title). Without a
|
||||||
|
// cache each episode re-issues the identical TMDb/Douban/Bangumi/TheTVDB
|
||||||
|
// search, so a 100-episode season costs 100x the network calls it needs.
|
||||||
|
//
|
||||||
|
// The cache lives on the ScraperService instance (not a package global) so that
|
||||||
|
// two services with different provider configuration never share results. TTL
|
||||||
|
// bounds staleness for out-of-band edits; explicit invalidation is unnecessary
|
||||||
|
// because the key includes the effective media kind, theatrical flag, query
|
||||||
|
// and year. Negative entries use a shorter TTL so a failed query is retried
|
||||||
|
// sooner without manual intervention.
|
||||||
|
const (
|
||||||
|
scrapeLookupCacheTTL = 10 * time.Minute
|
||||||
|
scrapeLookupNegativeCacheTTL = 2 * time.Minute
|
||||||
|
scrapeLookupCacheMaxItems = 1024
|
||||||
|
)
|
||||||
|
|
||||||
|
type scrapeLookupCache struct {
|
||||||
|
mu sync.Mutex
|
||||||
|
entries map[string]scrapeLookupCacheEntry
|
||||||
|
}
|
||||||
|
|
||||||
|
type scrapeLookupCacheEntry struct {
|
||||||
|
match *Match
|
||||||
|
expiresAt time.Time
|
||||||
|
}
|
||||||
|
|
||||||
|
func newScrapeLookupCache() *scrapeLookupCache {
|
||||||
|
return &scrapeLookupCache{entries: map[string]scrapeLookupCacheEntry{}}
|
||||||
|
}
|
||||||
|
|
||||||
|
func scrapeLookupCacheKey(kind, query string, year int, isTheatrical bool) string {
|
||||||
|
theatrical := "0"
|
||||||
|
if isTheatrical {
|
||||||
|
theatrical = "1"
|
||||||
|
}
|
||||||
|
return strings.ToLower(strings.TrimSpace(kind)) + "|" +
|
||||||
|
theatrical + "|" +
|
||||||
|
strconv.Itoa(year) + "|" +
|
||||||
|
strings.ToLower(strings.TrimSpace(query))
|
||||||
|
}
|
||||||
|
|
||||||
|
// get reports whether the key is cached. A hit may carry a nil match, which
|
||||||
|
// means the provider chain already ran and found nothing (negative cache).
|
||||||
|
// Use clearNegative before a user-triggered retry so stale negative entries
|
||||||
|
// do not suppress the fresh provider round-trip.
|
||||||
|
func (c *scrapeLookupCache) get(key string) (*Match, bool) {
|
||||||
|
if c == nil || key == "" {
|
||||||
|
return nil, false
|
||||||
|
}
|
||||||
|
now := time.Now()
|
||||||
|
c.mu.Lock()
|
||||||
|
defer c.mu.Unlock()
|
||||||
|
item, ok := c.entries[key]
|
||||||
|
if !ok {
|
||||||
|
return nil, false
|
||||||
|
}
|
||||||
|
if now.After(item.expiresAt) {
|
||||||
|
delete(c.entries, key)
|
||||||
|
return nil, false
|
||||||
|
}
|
||||||
|
return cloneMatch(item.match), true
|
||||||
|
}
|
||||||
|
|
||||||
|
func (c *scrapeLookupCache) set(key string, match *Match) {
|
||||||
|
if c == nil || key == "" {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
now := time.Now()
|
||||||
|
c.mu.Lock()
|
||||||
|
defer c.mu.Unlock()
|
||||||
|
if len(c.entries) >= scrapeLookupCacheMaxItems {
|
||||||
|
for k, item := range c.entries {
|
||||||
|
if now.After(item.expiresAt) || len(c.entries) >= scrapeLookupCacheMaxItems {
|
||||||
|
delete(c.entries, k)
|
||||||
|
}
|
||||||
|
if len(c.entries) < scrapeLookupCacheMaxItems {
|
||||||
|
break
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
ttl := scrapeLookupCacheTTL
|
||||||
|
if match == nil {
|
||||||
|
ttl = scrapeLookupNegativeCacheTTL
|
||||||
|
}
|
||||||
|
c.entries[key] = scrapeLookupCacheEntry{match: cloneMatch(match), expiresAt: now.Add(ttl)}
|
||||||
|
}
|
||||||
|
|
||||||
|
// clearNegatives drops cached misses so a user-triggered retry gets a fresh
|
||||||
|
// provider round-trip instead of reusing a stale negative entry.
|
||||||
|
func (c *scrapeLookupCache) clearNegatives() {
|
||||||
|
if c == nil {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
c.mu.Lock()
|
||||||
|
defer c.mu.Unlock()
|
||||||
|
for k, item := range c.entries {
|
||||||
|
if item.match == nil {
|
||||||
|
delete(c.entries, k)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// cloneMatch returns an independent copy so callers that mutate the result
|
||||||
|
// (localized title preference, local metadata merge, fanart artwork) cannot
|
||||||
|
// corrupt the cached entry or leak state between media rows.
|
||||||
|
func cloneMatch(m *Match) *Match {
|
||||||
|
if m == nil {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
out := *m
|
||||||
|
out.Languages = append([]string(nil), m.Languages...)
|
||||||
|
out.Countries = append([]string(nil), m.Countries...)
|
||||||
|
out.Genres = append([]string(nil), m.Genres...)
|
||||||
|
out.Aliases = append([]string(nil), m.Aliases...)
|
||||||
|
return &out
|
||||||
|
}
|
||||||
@@ -0,0 +1,165 @@
|
|||||||
|
package service
|
||||||
|
|
||||||
|
import (
|
||||||
|
"encoding/json"
|
||||||
|
"net/http"
|
||||||
|
"net/http/httptest"
|
||||||
|
"sync/atomic"
|
||||||
|
"testing"
|
||||||
|
|
||||||
|
"github.com/glebarez/sqlite"
|
||||||
|
"go.uber.org/zap"
|
||||||
|
"gorm.io/gorm"
|
||||||
|
|
||||||
|
"github.com/truewhile/MeBox/internal/config"
|
||||||
|
"github.com/truewhile/MeBox/internal/model"
|
||||||
|
"github.com/truewhile/MeBox/internal/repository"
|
||||||
|
)
|
||||||
|
|
||||||
|
// Every episode of a show produces the same query candidate (the series folder
|
||||||
|
// title). Without the lookup cache each episode re-issues the identical search,
|
||||||
|
// so a whole season costs one request per episode. This asserts the provider is
|
||||||
|
// hit once and subsequent episodes reuse the cached match.
|
||||||
|
func TestEnrichLibraryReusesLookupCacheAcrossEpisodes(t *testing.T) {
|
||||||
|
var searchCalls int32
|
||||||
|
upstream := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||||
|
w.Header().Set("Content-Type", "application/json")
|
||||||
|
switch r.URL.Path {
|
||||||
|
case "/search/tv":
|
||||||
|
if r.URL.Query().Get("query") != "折腰" {
|
||||||
|
_ = json.NewEncoder(w).Encode(map[string]any{"results": []any{}})
|
||||||
|
return
|
||||||
|
}
|
||||||
|
atomic.AddInt32(&searchCalls, 1)
|
||||||
|
_ = json.NewEncoder(w).Encode(map[string]any{
|
||||||
|
"results": []map[string]any{{
|
||||||
|
"id": 296753,
|
||||||
|
"name": "折腰",
|
||||||
|
"overview": "正确的剧集条目",
|
||||||
|
"poster_path": "/zheyao.jpg",
|
||||||
|
"first_air_date": "2025-05-13",
|
||||||
|
"origin_country": []string{"CN"},
|
||||||
|
}},
|
||||||
|
})
|
||||||
|
default:
|
||||||
|
http.NotFound(w, r)
|
||||||
|
}
|
||||||
|
}))
|
||||||
|
defer upstream.Close()
|
||||||
|
|
||||||
|
db, err := gorm.Open(sqlite.Open("file::memory:?cache=shared"), &gorm.Config{})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if err := db.AutoMigrate(&model.Library{}, &model.Series{}, &model.Media{}); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
repos := repository.New(db)
|
||||||
|
cfg := &config.Config{}
|
||||||
|
cfg.Secrets.TMDbAPIKey = "test-key"
|
||||||
|
cfg.Secrets.TMDbAPIProxy = upstream.URL
|
||||||
|
cfg.Secrets.TMDbImageProxy = upstream.URL + "/images"
|
||||||
|
log := zap.NewNop()
|
||||||
|
scraper := NewScraperService(cfg, log, repos, NewTMDbProvider(cfg, log, nil), nil, nil, nil, NewHub(log))
|
||||||
|
|
||||||
|
lib := model.Library{Name: "OpenList · 刮削缓存测试库", Path: "cloud://openlist/scrape-lookup-cache", Type: "tv", Enabled: true}
|
||||||
|
if err := repos.DB.Create(&lib).Error; err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
const episodeCount = 4
|
||||||
|
for episode := 1; episode <= episodeCount; episode++ {
|
||||||
|
media := model.Media{
|
||||||
|
LibraryID: lib.ID,
|
||||||
|
Title: "折腰",
|
||||||
|
Path: "cloud://openlist/scrape-lookup-cache/折腰 (2025)/Season 1/折腰.S01E0" + string(rune('0'+episode)) + ".mkv",
|
||||||
|
SeasonNum: 1,
|
||||||
|
EpisodeNum: episode,
|
||||||
|
ScrapeStatus: "pending",
|
||||||
|
}
|
||||||
|
if err := repos.DB.Create(&media).Error; err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
result, err := scraper.EnrichLibraryDetailedWithOptions(t.Context(), lib.ID, ScrapeOptions{})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if result.Matched != episodeCount {
|
||||||
|
t.Fatalf("matched = %d, want %d", result.Matched, episodeCount)
|
||||||
|
}
|
||||||
|
if calls := atomic.LoadInt32(&searchCalls); calls != 1 {
|
||||||
|
t.Fatalf("tmdb /search/tv called %d times for %d episodes; want 1 (cache should dedupe identical queries)", calls, episodeCount)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// A cached match must be handed out as an independent copy: mutating one row's
|
||||||
|
// result (localized title preference, local metadata merge) must not bleed into
|
||||||
|
// the next row.
|
||||||
|
func TestScrapeLookupCacheReturnsIndependentCopies(t *testing.T) {
|
||||||
|
cache := newScrapeLookupCache()
|
||||||
|
original := &Match{
|
||||||
|
Title: "折腰",
|
||||||
|
TMDbID: 296753,
|
||||||
|
Genres: []string{"剧情"},
|
||||||
|
Countries: []string{"CN"},
|
||||||
|
Aliases: []string{"Zhe Yao"},
|
||||||
|
}
|
||||||
|
key := scrapeLookupCacheKey("tv", "折腰", 2025, false)
|
||||||
|
cache.set(key, original)
|
||||||
|
|
||||||
|
first, ok := cache.get(key)
|
||||||
|
if !ok || first == nil {
|
||||||
|
t.Fatal("expected a cache hit")
|
||||||
|
}
|
||||||
|
first.Title = "mutated"
|
||||||
|
first.Genres[0] = "mutated"
|
||||||
|
first.Aliases = append(first.Aliases, "extra")
|
||||||
|
|
||||||
|
second, ok := cache.get(key)
|
||||||
|
if !ok || second == nil {
|
||||||
|
t.Fatal("expected a second cache hit")
|
||||||
|
}
|
||||||
|
if second.Title != "折腰" || second.Genres[0] != "剧情" || len(second.Aliases) != 1 {
|
||||||
|
t.Fatalf("cached match was mutated through a returned copy: %+v", second)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Negative results are cached too: a query that matched nothing must not be
|
||||||
|
// re-issued for every remaining episode of the same show.
|
||||||
|
func TestScrapeLookupCacheStoresNegativeResults(t *testing.T) {
|
||||||
|
cache := newScrapeLookupCache()
|
||||||
|
key := scrapeLookupCacheKey("tv", "no-such-show", 0, false)
|
||||||
|
cache.set(key, nil)
|
||||||
|
if _, ok := cache.get(key); !ok {
|
||||||
|
t.Fatal("negative result should be cached to avoid repeated provider calls")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// The theatrical flag is part of the key: a theatrical feature gets a tv
|
||||||
|
// fallback lookup, so its result must not be shared with (or returned for)
|
||||||
|
// the same query issued for a non-theatrical media row.
|
||||||
|
func TestScrapeLookupCacheKeySeparatesTheatrical(t *testing.T) {
|
||||||
|
plain := scrapeLookupCacheKey("movie", "query", 2024, false)
|
||||||
|
theatrical := scrapeLookupCacheKey("movie", "query", 2024, true)
|
||||||
|
if plain == theatrical {
|
||||||
|
t.Fatalf("theatrical flag must be part of the cache key: %q", plain)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// A manual "retry no match" run clears stale negative entries so the provider
|
||||||
|
// chain is actually re-queried instead of short-circuiting on the cached miss.
|
||||||
|
func TestScrapeLookupCacheClearNegativesKeepsPositives(t *testing.T) {
|
||||||
|
cache := newScrapeLookupCache()
|
||||||
|
negKey := scrapeLookupCacheKey("tv", "missing", 0, false)
|
||||||
|
posKey := scrapeLookupCacheKey("tv", "found", 0, false)
|
||||||
|
cache.set(negKey, nil)
|
||||||
|
cache.set(posKey, &Match{Title: "found"})
|
||||||
|
cache.clearNegatives()
|
||||||
|
if _, ok := cache.get(negKey); ok {
|
||||||
|
t.Fatal("negative entry should be cleared before a manual retry")
|
||||||
|
}
|
||||||
|
if got, ok := cache.get(posKey); !ok || got == nil || got.Title != "found" {
|
||||||
|
t.Fatalf("positive entry should survive clearNegatives: %+v", got)
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -24,11 +24,15 @@ var noiseTokens = []string{
|
|||||||
"hkfree", "yify", "rarbg", "ettv", "fgt", "tgx", "ctrlhd", "ntb", "flux", "qhstudio",
|
"hkfree", "yify", "rarbg", "ettv", "fgt", "tgx", "ctrlhd", "ntb", "flux", "qhstudio",
|
||||||
|
|
||||||
// 流媒体平台 / 字幕组 / 国家版本(动漫常见)
|
// 流媒体平台 / 字幕组 / 国家版本(动漫常见)
|
||||||
"netflix", "nf", "amzn", "hulu", "disney", "max", "hbo",
|
// 注意:不要把同时是常见英文单词的标记放进来(如 max / web / judas),
|
||||||
|
// 否则 "Mad Max"、"Web Therapy" 这类正常标题会被误删。裸 "WEB" 发布标记
|
||||||
|
// 通常位于分辨率之后,由 releaseBoundary 截断规则处理;"WEB-DL" 则由
|
||||||
|
// multiWordNoise 单独匹配。
|
||||||
|
"netflix", "nf", "amzn", "hulu", "disney", "hbo",
|
||||||
"linetv", "ourtv", "iqiyi", "youku", "bilibili", "qiyi", "krj",
|
"linetv", "ourtv", "iqiyi", "youku", "bilibili", "qiyi", "krj",
|
||||||
"atvp", "appletv", "apple-tv", "tx", "txweb",
|
"atvp", "appletv", "apple-tv", "tx", "txweb",
|
||||||
"crunchyroll", "funimation", "anidb", "horriblesubs", "subsplease",
|
"crunchyroll", "funimation", "anidb", "horriblesubs", "subsplease",
|
||||||
"erai-raws", "judas", "asw", "smcat", "leopard-raws", "ohys-raws", "colortv",
|
"erai-raws", "asw", "smcat", "leopard-raws", "ohys-raws", "colortv",
|
||||||
"mweb", "ubweb", "hhweb", "adweb", "chdweb", "kurosawa", "qhstudio",
|
"mweb", "ubweb", "hhweb", "adweb", "chdweb", "kurosawa", "qhstudio",
|
||||||
|
|
||||||
// 中文字幕标记
|
// 中文字幕标记
|
||||||
@@ -55,6 +59,20 @@ var releaseBoundaryTokenSet = map[string]struct{}{
|
|||||||
"x264": {}, "x265": {}, "h264": {}, "h265": {}, "h266": {}, "hevc": {}, "avc": {}, "av1": {}, "vvc": {},
|
"x264": {}, "x265": {}, "h264": {}, "h265": {}, "h266": {}, "hevc": {}, "avc": {}, "av1": {}, "vvc": {},
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// weakReleaseBoundaryTokenSet are release tags that double as plausible title
|
||||||
|
// words. Before any release signal has been seen they are kept as part of the
|
||||||
|
// title ("Mad Max", "Web Therapy"), so a title is never truncated to nothing;
|
||||||
|
// after a real signal they behave like any other tag.
|
||||||
|
var weakReleaseBoundaryTokenSet = map[string]struct{}{
|
||||||
|
"bd": {}, "dvd": {}, "web": {},
|
||||||
|
}
|
||||||
|
|
||||||
|
// releaseSignalToken marks the position of an extracted year. The year itself
|
||||||
|
// is not a title token, but its presence still proves the following tokens are
|
||||||
|
// release tags ("复仇者联盟4.2019.BD.1080p" must drop "BD"). A control rune is
|
||||||
|
// used so it can never collide with a real filename token.
|
||||||
|
const releaseSignalToken = "\u0001"
|
||||||
|
|
||||||
var dynamicReleaseBoundaryTokenRE = regexp.MustCompile(`(?i)^(?:\d{3,4}p|\d{2,3}fps)$`)
|
var dynamicReleaseBoundaryTokenRE = regexp.MustCompile(`(?i)^(?:\d{3,4}p|\d{2,3}fps)$`)
|
||||||
|
|
||||||
// bracketedTag matches "[anything]", "(anything)" or "{anything}" segments.
|
// bracketedTag matches "[anything]", "(anything)" or "{anything}" segments.
|
||||||
@@ -86,7 +104,9 @@ func CleanQuery(raw string) (title string, year int) {
|
|||||||
if m := yearPattern.FindStringSubmatch(lower); len(m) >= 2 {
|
if m := yearPattern.FindStringSubmatch(lower); len(m) >= 2 {
|
||||||
if v, err := strconv.Atoi(m[1]); err == nil {
|
if v, err := strconv.Atoi(m[1]); err == nil {
|
||||||
year = v
|
year = v
|
||||||
lower = strings.ReplaceAll(lower, m[1], " ")
|
// Keep a positional marker: the year is not a title token, but its
|
||||||
|
// presence arms release-tag truncation for what follows.
|
||||||
|
lower = strings.ReplaceAll(lower, m[1], " "+releaseSignalToken+" ")
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -110,17 +130,35 @@ func CleanQuery(raw string) (title string, year int) {
|
|||||||
}
|
}
|
||||||
// 拆分后丢掉过短(≤1)且全为 ASCII 数字 / 字母的"碎片",避免
|
// 拆分后丢掉过短(≤1)且全为 ASCII 数字 / 字母的"碎片",避免
|
||||||
// 「2」「0」「v」之类残留干扰 TMDb 搜索。中文字符不算碎片。
|
// 「2」「0」「v」之类残留干扰 TMDb 搜索。中文字符不算碎片。
|
||||||
|
//
|
||||||
|
// seenReleaseBoundary 只有在遇到分辨率/编码等强标记后才置位,置位后其余
|
||||||
|
// ASCII 词一律视为发布尾巴丢弃;releaseSignalled 则宽松得多,年份标记也会
|
||||||
|
// 置位,它只用来武装弱标记(bd/dvd/web)。这样 "Web Therapy" 不会被截空,
|
||||||
|
// 而 "2019.Avatar.1080p" 这种年份前置的标题也不会因为年份标记把后面的
|
||||||
|
// 真实标题词当成尾巴丢掉。
|
||||||
out := make([]string, 0, 8)
|
out := make([]string, 0, 8)
|
||||||
seenReleaseBoundary := false
|
seenReleaseBoundary := false
|
||||||
|
releaseSignalled := false
|
||||||
for _, w := range strings.Fields(lower) {
|
for _, w := range strings.Fields(lower) {
|
||||||
|
if w == releaseSignalToken {
|
||||||
|
releaseSignalled = true
|
||||||
|
continue
|
||||||
|
}
|
||||||
if dynamicReleaseBoundaryTokenRE.MatchString(w) {
|
if dynamicReleaseBoundaryTokenRE.MatchString(w) {
|
||||||
|
releaseSignalled = true
|
||||||
seenReleaseBoundary = true
|
seenReleaseBoundary = true
|
||||||
continue
|
continue
|
||||||
}
|
}
|
||||||
if _, ok := noiseTokenSet[w]; ok {
|
if _, boundary := releaseBoundaryTokenSet[w]; boundary {
|
||||||
if _, boundary := releaseBoundaryTokenSet[w]; boundary {
|
if _, weak := weakReleaseBoundaryTokenSet[w]; weak && !releaseSignalled {
|
||||||
|
// Ambiguous tag that is also a plausible title word; keep it
|
||||||
|
// until a real release signal proves we are past the title.
|
||||||
|
} else {
|
||||||
|
releaseSignalled = true
|
||||||
seenReleaseBoundary = true
|
seenReleaseBoundary = true
|
||||||
|
continue
|
||||||
}
|
}
|
||||||
|
} else if _, ok := noiseTokenSet[w]; ok {
|
||||||
continue
|
continue
|
||||||
}
|
}
|
||||||
if seenReleaseBoundary && isASCIIWord(w) {
|
if seenReleaseBoundary && isASCIIWord(w) {
|
||||||
|
|||||||
@@ -0,0 +1,77 @@
|
|||||||
|
package service
|
||||||
|
|
||||||
|
import "testing"
|
||||||
|
|
||||||
|
// CleanQuery must not delete words that are also ordinary English title words
|
||||||
|
// just because a release group reused them as a tag. Regression guard: "max"
|
||||||
|
// and "web" used to be unconditional noise/boundary tokens, which turned
|
||||||
|
// "Mad Max" into "mad" and "Web Therapy" into an empty query.
|
||||||
|
func TestCleanQueryKeepsCommonEnglishTitleWords(t *testing.T) {
|
||||||
|
cases := []struct {
|
||||||
|
in string
|
||||||
|
wantTitle string
|
||||||
|
wantYear int
|
||||||
|
}{
|
||||||
|
{"Mad.Max.1979.1080p.BluRay.mkv", "mad max", 1979},
|
||||||
|
{"Max.Payne.2008.1080p.WEB-DL.mkv", "max payne", 2008},
|
||||||
|
{"Web.Therapy.S01E01.1080p.WEB-DL.mkv", "web therapy", 0},
|
||||||
|
{"The.Web.2019.1080p.mkv", "the web", 2019},
|
||||||
|
}
|
||||||
|
for _, tc := range cases {
|
||||||
|
t.Run(tc.in, func(t *testing.T) {
|
||||||
|
gotTitle, gotYear := CleanQuery(tc.in)
|
||||||
|
if gotTitle != tc.wantTitle || gotYear != tc.wantYear {
|
||||||
|
t.Errorf("CleanQuery(%q) = (%q, %d), want (%q, %d)",
|
||||||
|
tc.in, gotTitle, gotYear, tc.wantTitle, tc.wantYear)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// A title consisting only of an ambiguous tag plus a release tail must not be
|
||||||
|
// truncated to nothing; the tag stays so the query still has a chance.
|
||||||
|
func TestCleanQueryDoesNotTruncateAmbiguousTagToNothing(t *testing.T) {
|
||||||
|
for _, in := range []string{"Web.1080p.WEB-DL.mkv", "BD.720p.mkv"} {
|
||||||
|
title, _ := CleanQuery(in)
|
||||||
|
if title == "" {
|
||||||
|
t.Errorf("CleanQuery(%q) returned an empty title", in)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// A year-prefixed filename must keep its title: the year arms weak-tag
|
||||||
|
// truncation but must not discard the real title words that follow it.
|
||||||
|
func TestCleanQueryKeepsTitleAfterYearPrefix(t *testing.T) {
|
||||||
|
cases := map[string]string{
|
||||||
|
"2019.Avatar.1080p.BluRay.mkv": "avatar",
|
||||||
|
"2024.Dune.Part.Two.2160p.WEB.mkv": "dune part two",
|
||||||
|
}
|
||||||
|
for in, want := range cases {
|
||||||
|
t.Run(in, func(t *testing.T) {
|
||||||
|
got, year := CleanQuery(in)
|
||||||
|
if got != want {
|
||||||
|
t.Errorf("CleanQuery(%q) = (%q, %d), want title %q", in, got, year, want)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Release-tag truncation must still fire once a real release signal (an
|
||||||
|
// extracted year, resolution or codec) has been seen, so tags after it are
|
||||||
|
// dropped as before.
|
||||||
|
func TestCleanQueryStillDropsTagsAfterReleaseSignal(t *testing.T) {
|
||||||
|
cases := map[string]string{
|
||||||
|
"复仇者联盟4.2019.BD.1080p.mkv": "复仇者联盟4",
|
||||||
|
"The.Matrix.1999.1080p.WEB-DL.H265.mp4": "the matrix",
|
||||||
|
"Oppenheimer.2023.2160p.UHD.BluRay.mkv": "oppenheimer",
|
||||||
|
"Interstellar.2014.4k.hdr.dts.atmos.mkv": "interstellar",
|
||||||
|
}
|
||||||
|
for in, want := range cases {
|
||||||
|
t.Run(in, func(t *testing.T) {
|
||||||
|
got, _ := CleanQuery(in)
|
||||||
|
if got != want {
|
||||||
|
t.Errorf("CleanQuery(%q) = %q, want %q", in, got, want)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -26,6 +26,10 @@ type ScraperService struct {
|
|||||||
cache *RuntimeCacheService
|
cache *RuntimeCacheService
|
||||||
images *ImageProxy
|
images *ImageProxy
|
||||||
|
|
||||||
|
// Caches provider-chain results keyed by media kind/query/year. Per-instance
|
||||||
|
// so services with different provider config never share results.
|
||||||
|
lookupCache *scrapeLookupCache
|
||||||
|
|
||||||
// Serializes final sidecar replacement. Windows cannot rename over an
|
// Serializes final sidecar replacement. Windows cannot rename over an
|
||||||
// existing file, and concurrent scrapes can target the same sidecar.
|
// existing file, and concurrent scrapes can target the same sidecar.
|
||||||
artworkWriteMu sync.Mutex
|
artworkWriteMu sync.Mutex
|
||||||
@@ -50,6 +54,7 @@ func NewScraperService(
|
|||||||
return &ScraperService{
|
return &ScraperService{
|
||||||
cfg: cfg, log: log, repo: repo,
|
cfg: cfg, log: log, repo: repo,
|
||||||
tmdb: tmdb, bangumi: bangumi, thetvdb: thetvdb, fanart: fanart, adult: adultProvider, hub: hub,
|
tmdb: tmdb, bangumi: bangumi, thetvdb: thetvdb, fanart: fanart, adult: adultProvider, hub: hub,
|
||||||
|
lookupCache: newScrapeLookupCache(),
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user