mirror of
https://github.com/truewhile/MeBox.git
synced 2026-09-29 11:36:36 +08:00
优化
This commit is contained in:
@@ -55,7 +55,7 @@ func (o *OrganizerService) lookupOrganizeMetadata(ctx context.Context, src, sour
|
||||
continue
|
||||
}
|
||||
}
|
||||
match := o.scraper.lookup(ctx, lib, media, candidate, year)
|
||||
match := o.scraper.lookup(ctx, lib, media, candidate, year, false)
|
||||
if match != nil && strings.TrimSpace(match.Title) != "" {
|
||||
if !organizeMetadataMatchTrusted(candidate, year, match) {
|
||||
if cache != nil {
|
||||
|
||||
@@ -5,8 +5,10 @@ import (
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"net/http"
|
||||
"regexp"
|
||||
"strconv"
|
||||
"strings"
|
||||
"sync"
|
||||
"time"
|
||||
|
||||
"go.uber.org/zap"
|
||||
@@ -61,8 +63,9 @@ func NewRecognitionWordsService(log *zap.Logger, repo *repository.Container) *Re
|
||||
}
|
||||
|
||||
func (s *RecognitionWordsService) Config(ctx context.Context) RecognitionWordsConfig {
|
||||
cfg := recognitionWordsConfig(ctx, s.repo)
|
||||
cfg.RuleCount = len(parseRecognitionWordRules(recognitionWordsCombinedText(cfg)))
|
||||
cfg, rules := recognitionWordsRuntime(ctx, s.repo)
|
||||
cfg = cloneRecognitionWordsConfig(cfg)
|
||||
cfg.RuleCount = len(rules)
|
||||
return cfg
|
||||
}
|
||||
|
||||
@@ -80,7 +83,11 @@ func (s *RecognitionWordsService) SaveConfig(ctx context.Context, cfg Recognitio
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
return s.repo.Setting.Set(ctx, RecognitionWordsSharedURLsKey, string(rawURLs))
|
||||
if err := s.repo.Setting.Set(ctx, RecognitionWordsSharedURLsKey, string(rawURLs)); err != nil {
|
||||
return err
|
||||
}
|
||||
invalidateRecognitionWordsRuntime(s.repo)
|
||||
return nil
|
||||
}
|
||||
|
||||
func (s *RecognitionWordsService) SyncShared(ctx context.Context) (RecognitionWordsConfig, error) {
|
||||
@@ -104,6 +111,7 @@ func (s *RecognitionWordsService) SyncShared(ctx context.Context) (RecognitionWo
|
||||
if err := s.repo.Setting.Set(ctx, RecognitionWordsSyncedAtKey, now); err != nil {
|
||||
return cfg, err
|
||||
}
|
||||
invalidateRecognitionWordsRuntime(s.repo)
|
||||
return s.Config(ctx), nil
|
||||
}
|
||||
|
||||
@@ -120,11 +128,10 @@ func (s *RecognitionWordsService) Test(ctx context.Context, input string) Recogn
|
||||
}
|
||||
|
||||
func ApplyRecognitionWords(ctx context.Context, repo *repository.Container, raw string) string {
|
||||
cfg := recognitionWordsConfig(ctx, repo)
|
||||
if !cfg.Enabled {
|
||||
cfg, rules := recognitionWordsRuntime(ctx, repo)
|
||||
if !cfg.Enabled || len(rules) == 0 {
|
||||
return raw
|
||||
}
|
||||
rules := parseRecognitionWordRules(recognitionWordsCombinedText(cfg))
|
||||
return applyRecognitionWordRules(raw, rules)
|
||||
}
|
||||
|
||||
@@ -189,9 +196,69 @@ func normalizeRecognitionWordURLs(values []string) []string {
|
||||
type recognitionWordRule struct {
|
||||
raw string
|
||||
block string
|
||||
blockRE *regexp.Regexp
|
||||
replaceFrom string
|
||||
replaceTo string
|
||||
replaceRE *regexp.Regexp
|
||||
offsetLeft string
|
||||
offsetRight string
|
||||
offsetExpr string
|
||||
offsetRE *regexp.Regexp
|
||||
}
|
||||
|
||||
// recognitionWordsRuntime caches the parsed rule set (with pre-compiled
|
||||
// regexes) per repository. Reading five settings rows and parsing/recompiling
|
||||
// the whole shared word list used to happen on every query candidate, which is
|
||||
// the scraper's hottest path. Writes through SaveConfig/SyncShared invalidate
|
||||
// the entry immediately; the TTL only bounds staleness for out-of-process edits.
|
||||
type recognitionWordsRuntimeEntry struct {
|
||||
repo *repository.Container
|
||||
cfg RecognitionWordsConfig
|
||||
rules []recognitionWordRule
|
||||
expiresAt time.Time
|
||||
}
|
||||
|
||||
const recognitionWordsRuntimeTTL = 30 * time.Second
|
||||
|
||||
var recognitionWordsRuntimeCache struct {
|
||||
sync.RWMutex
|
||||
entry recognitionWordsRuntimeEntry
|
||||
valid bool
|
||||
}
|
||||
|
||||
func recognitionWordsRuntime(ctx context.Context, repo *repository.Container) (RecognitionWordsConfig, []recognitionWordRule) {
|
||||
now := time.Now()
|
||||
recognitionWordsRuntimeCache.RLock()
|
||||
cached := recognitionWordsRuntimeCache.entry
|
||||
hit := recognitionWordsRuntimeCache.valid && cached.repo == repo && now.Before(cached.expiresAt)
|
||||
recognitionWordsRuntimeCache.RUnlock()
|
||||
if hit {
|
||||
return cached.cfg, cached.rules
|
||||
}
|
||||
|
||||
cfg := recognitionWordsConfig(ctx, repo)
|
||||
rules := parseRecognitionWordRules(recognitionWordsCombinedText(cfg))
|
||||
recognitionWordsRuntimeCache.Lock()
|
||||
recognitionWordsRuntimeCache.entry = recognitionWordsRuntimeEntry{
|
||||
repo: repo,
|
||||
cfg: cfg,
|
||||
rules: rules,
|
||||
expiresAt: now.Add(recognitionWordsRuntimeTTL),
|
||||
}
|
||||
recognitionWordsRuntimeCache.valid = true
|
||||
recognitionWordsRuntimeCache.Unlock()
|
||||
return cfg, rules
|
||||
}
|
||||
|
||||
func invalidateRecognitionWordsRuntime(repo *repository.Container) {
|
||||
recognitionWordsRuntimeCache.Lock()
|
||||
if recognitionWordsRuntimeCache.entry.repo == repo {
|
||||
recognitionWordsRuntimeCache.valid = false
|
||||
}
|
||||
recognitionWordsRuntimeCache.Unlock()
|
||||
}
|
||||
|
||||
func cloneRecognitionWordsConfig(cfg RecognitionWordsConfig) RecognitionWordsConfig {
|
||||
cfg.SharedURLs = append([]string(nil), cfg.SharedURLs...)
|
||||
return cfg
|
||||
}
|
||||
|
||||
@@ -0,0 +1,49 @@
|
||||
package service
|
||||
|
||||
import (
|
||||
"testing"
|
||||
|
||||
"go.uber.org/zap"
|
||||
)
|
||||
|
||||
// The rule set is cached process-wide, so a config write must invalidate it
|
||||
// immediately; otherwise the admin would have to wait out the TTL before a
|
||||
// saved word list takes effect.
|
||||
func TestRecognitionWordsCacheInvalidatedBySaveConfig(t *testing.T) {
|
||||
repos := newOrganizerTestRepo(t)
|
||||
svc := NewRecognitionWordsService(zap.NewNop(), repos)
|
||||
|
||||
if err := svc.SaveConfig(t.Context(), RecognitionWordsConfig{
|
||||
Enabled: true,
|
||||
LocalText: "BADWORD => 好标题",
|
||||
}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if got := ApplyRecognitionWords(t.Context(), repos, "BADWORD"); got != "好标题" {
|
||||
t.Fatalf("first apply = %q, want 好标题", got)
|
||||
}
|
||||
|
||||
if err := svc.SaveConfig(t.Context(), RecognitionWordsConfig{
|
||||
Enabled: true,
|
||||
LocalText: "BADWORD => 新标题",
|
||||
}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if got := ApplyRecognitionWords(t.Context(), repos, "BADWORD"); got != "新标题" {
|
||||
t.Fatalf("cached rules were not invalidated after SaveConfig: got %q, want 新标题", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRecognitionWordsDisabledReturnsRawInput(t *testing.T) {
|
||||
repos := newOrganizerTestRepo(t)
|
||||
svc := NewRecognitionWordsService(zap.NewNop(), repos)
|
||||
if err := svc.SaveConfig(t.Context(), RecognitionWordsConfig{
|
||||
Enabled: false,
|
||||
LocalText: "BADWORD => 好标题",
|
||||
}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if got := ApplyRecognitionWords(t.Context(), repos, "BADWORD"); got != "BADWORD" {
|
||||
t.Fatalf("disabled recognition words changed input: got %q", got)
|
||||
}
|
||||
}
|
||||
@@ -31,6 +31,11 @@ func parseRecognitionWordRule(line string) recognitionWordRule {
|
||||
case strings.Contains(part, "<>") && strings.Contains(part, ">>"):
|
||||
beforeAfter := strings.SplitN(part, ">>", 2)
|
||||
bounds := strings.SplitN(beforeAfter[0], "<>", 2)
|
||||
// A malformed rule without "<>" would otherwise index past the
|
||||
// slice; word lists are fetched from the network, so stay defensive.
|
||||
if len(bounds) < 2 {
|
||||
continue
|
||||
}
|
||||
rule.offsetLeft = strings.TrimSpace(bounds[0])
|
||||
rule.offsetRight = strings.TrimSpace(bounds[1])
|
||||
rule.offsetExpr = strings.TrimSpace(beforeAfter[1])
|
||||
@@ -38,56 +43,89 @@ func parseRecognitionWordRule(line string) recognitionWordRule {
|
||||
rule.block = part
|
||||
}
|
||||
}
|
||||
compileRecognitionWordRule(&rule)
|
||||
return rule
|
||||
}
|
||||
|
||||
var recognitionReplacementRE = regexp.MustCompile(`\\([0-9]+)`)
|
||||
|
||||
func normalizeRecognitionReplacement(value string) string {
|
||||
re := regexp.MustCompile(`\\([0-9]+)`)
|
||||
return re.ReplaceAllString(value, "$$$1")
|
||||
return recognitionReplacementRE.ReplaceAllString(value, "$$$1")
|
||||
}
|
||||
|
||||
// compileRecognitionWordRule pre-compiles every pattern in a rule so the hot
|
||||
// clean-query path never recompiles regexes per candidate.
|
||||
func compileRecognitionWordRule(rule *recognitionWordRule) {
|
||||
if rule == nil {
|
||||
return
|
||||
}
|
||||
if rule.block != "" {
|
||||
if re, err := regexp.Compile(rule.block); err == nil {
|
||||
rule.blockRE = re
|
||||
}
|
||||
}
|
||||
if rule.replaceFrom != "" {
|
||||
if re, err := regexp.Compile(rule.replaceFrom); err == nil {
|
||||
rule.replaceRE = re
|
||||
}
|
||||
}
|
||||
if rule.offsetExpr != "" && (rule.offsetLeft != "" || rule.offsetRight != "") {
|
||||
if re, err := compileRecognitionOffsetRE(rule.offsetLeft, rule.offsetRight); err == nil {
|
||||
rule.offsetRE = re
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func compileRecognitionOffsetRE(left, right string) (*regexp.Regexp, error) {
|
||||
leftPattern := firstNonEmpty(left, `^`)
|
||||
rightPattern := firstNonEmpty(right, `$`)
|
||||
return regexp.Compile(`(?i)(` + leftPattern + `)(\d{1,5})(` + rightPattern + `)`)
|
||||
}
|
||||
|
||||
func applyRecognitionWordRules(raw string, rules []recognitionWordRule) string {
|
||||
out := strings.TrimSpace(raw)
|
||||
for _, rule := range rules {
|
||||
if rule.block != "" {
|
||||
out = applyRecognitionBlock(out, rule.block)
|
||||
out = applyRecognitionBlock(out, rule)
|
||||
}
|
||||
if rule.replaceFrom != "" {
|
||||
out = applyRecognitionReplace(out, rule.replaceFrom, rule.replaceTo)
|
||||
out = applyRecognitionReplace(out, rule)
|
||||
}
|
||||
if rule.offsetLeft != "" || rule.offsetRight != "" {
|
||||
out = applyRecognitionOffset(out, rule.offsetLeft, rule.offsetRight, rule.offsetExpr)
|
||||
out = applyRecognitionOffset(out, rule)
|
||||
}
|
||||
}
|
||||
return strings.Join(strings.Fields(out), " ")
|
||||
}
|
||||
|
||||
func applyRecognitionBlock(raw, block string) string {
|
||||
if re, err := regexp.Compile(block); err == nil {
|
||||
return re.ReplaceAllString(raw, " ")
|
||||
func applyRecognitionBlock(raw string, rule recognitionWordRule) string {
|
||||
if rule.blockRE != nil {
|
||||
return rule.blockRE.ReplaceAllString(raw, " ")
|
||||
}
|
||||
return strings.ReplaceAll(raw, block, " ")
|
||||
return strings.ReplaceAll(raw, rule.block, " ")
|
||||
}
|
||||
|
||||
func applyRecognitionReplace(raw, from, to string) string {
|
||||
if re, err := regexp.Compile(from); err == nil {
|
||||
return re.ReplaceAllString(raw, to)
|
||||
func applyRecognitionReplace(raw string, rule recognitionWordRule) string {
|
||||
if rule.replaceRE != nil {
|
||||
return rule.replaceRE.ReplaceAllString(raw, rule.replaceTo)
|
||||
}
|
||||
return strings.ReplaceAll(raw, from, to)
|
||||
return strings.ReplaceAll(raw, rule.replaceFrom, rule.replaceTo)
|
||||
}
|
||||
|
||||
func applyRecognitionOffset(raw, left, right, expr string) string {
|
||||
if strings.TrimSpace(expr) == "" {
|
||||
func applyRecognitionOffset(raw string, rule recognitionWordRule) string {
|
||||
if strings.TrimSpace(rule.offsetExpr) == "" {
|
||||
return raw
|
||||
}
|
||||
leftPattern := firstNonEmpty(left, `^`)
|
||||
rightPattern := firstNonEmpty(right, `$`)
|
||||
re, err := regexp.Compile(`(?i)(` + leftPattern + `)(\d{1,5})(` + rightPattern + `)`)
|
||||
if err != nil {
|
||||
return raw
|
||||
re := rule.offsetRE
|
||||
if re == nil {
|
||||
compiled, err := compileRecognitionOffsetRE(rule.offsetLeft, rule.offsetRight)
|
||||
if err != nil {
|
||||
return raw
|
||||
}
|
||||
re = compiled
|
||||
}
|
||||
return re.ReplaceAllStringFunc(raw, func(match string) string {
|
||||
return applyRecognitionOffsetMatch(re, match, expr)
|
||||
return applyRecognitionOffsetMatch(re, match, rule.offsetExpr)
|
||||
})
|
||||
}
|
||||
|
||||
|
||||
@@ -63,12 +63,17 @@ func (s *ScraperService) EnrichOneWithOptions(ctx context.Context, m *model.Medi
|
||||
}
|
||||
}
|
||||
|
||||
// Same stale-negative problem as the library path: a single-item retry
|
||||
// (queue task / manual rescrape) must get a fresh provider round-trip.
|
||||
if options.RetryNoMatch && s != nil {
|
||||
s.lookupCache.clearNegatives()
|
||||
}
|
||||
candidates := scrapeQueryCandidatesWithRecognition(ctx, s.repo, m, lib)
|
||||
var query string
|
||||
match := (*Match)(nil)
|
||||
for _, candidate := range candidates {
|
||||
query = candidate
|
||||
candidateMatch := s.lookup(ctx, lib, m, candidate, year)
|
||||
candidateMatch := s.lookup(ctx, lib, m, candidate, year, options.ForceRematch)
|
||||
if candidateMatch == nil {
|
||||
continue
|
||||
}
|
||||
|
||||
@@ -13,7 +13,12 @@ import (
|
||||
// lookup runs the provider chain after local NFO has been considered:
|
||||
// TMDb -> Douban -> Bangumi -> TheTVDB. Douban and Bangumi do not require API
|
||||
// keys; providers that are unavailable or return an error are skipped.
|
||||
func (s *ScraperService) lookup(ctx context.Context, lib *model.Library, media *model.Media, query string, year int) *Match {
|
||||
//
|
||||
// Results are cached per (kind, theatrical, query, year) because every
|
||||
// episode of a show produces the same candidate; bypassCache forces a fresh
|
||||
// provider round-trip for user-triggered rematches but still refreshes the
|
||||
// cache so later episodes reuse the corrected result.
|
||||
func (s *ScraperService) lookup(ctx context.Context, lib *model.Library, media *model.Media, query string, year int, bypassCache bool) *Match {
|
||||
kind := ""
|
||||
if lib != nil {
|
||||
kind = lib.Type
|
||||
@@ -28,6 +33,21 @@ func (s *ScraperService) lookup(ctx context.Context, lib *model.Library, media *
|
||||
} else if (normalizeOrganizeMediaType(kind) != "movie" || explicitEpisode) && mediaIsEpisodic(media, lib) {
|
||||
kind = "tv"
|
||||
}
|
||||
|
||||
cacheKey := scrapeLookupCacheKey(kind, query, year, isTheatrical)
|
||||
if !bypassCache {
|
||||
if cached, ok := s.lookupCache.get(cacheKey); ok {
|
||||
return cached
|
||||
}
|
||||
}
|
||||
match := s.lookupUncached(ctx, kind, isTheatrical, query, year)
|
||||
// Always refresh the cache, even on bypassed (forced) lookups, so a
|
||||
// corrected rematch replaces the stale entry for later episodes.
|
||||
s.lookupCache.set(cacheKey, match)
|
||||
return match
|
||||
}
|
||||
|
||||
func (s *ScraperService) lookupUncached(ctx context.Context, kind string, isTheatrical bool, query string, year int) *Match {
|
||||
if s.tmdb != nil && s.tmdb.Enabled() {
|
||||
if match := s.lookupAutomaticTMDb(ctx, kind, query, year); match != nil {
|
||||
match.Provider = "tmdb"
|
||||
@@ -107,6 +127,13 @@ func (s *ScraperService) EnrichLibraryDetailed(ctx context.Context, libraryID st
|
||||
|
||||
func (s *ScraperService) EnrichLibraryDetailedWithOptions(ctx context.Context, libraryID string, options ScrapeOptions) (EnrichLibraryResult, error) {
|
||||
result := EnrichLibraryResult{LibraryID: libraryID}
|
||||
// A manual retry must not reuse stale negative cache entries from the
|
||||
// previous run, otherwise "重新刮削" looks like it did nothing. Positives
|
||||
// are kept so the rest of the season still dedupes; the first re-queried
|
||||
// episode refreshes the entry for the rows behind it.
|
||||
if options.RetryNoMatch && s != nil {
|
||||
s.lookupCache.clearNegatives()
|
||||
}
|
||||
rows, err := s.scrapeCandidateRows(ctx, libraryID, options)
|
||||
if err != nil {
|
||||
return result, err
|
||||
|
||||
@@ -0,0 +1,126 @@
|
||||
package service
|
||||
|
||||
import (
|
||||
"strconv"
|
||||
"strings"
|
||||
"sync"
|
||||
"time"
|
||||
)
|
||||
|
||||
// A TV library scrapes one media row per episode, and every row in the same
|
||||
// show computes the same query candidate (the series folder title). Without a
|
||||
// cache each episode re-issues the identical TMDb/Douban/Bangumi/TheTVDB
|
||||
// search, so a 100-episode season costs 100x the network calls it needs.
|
||||
//
|
||||
// The cache lives on the ScraperService instance (not a package global) so that
|
||||
// two services with different provider configuration never share results. TTL
|
||||
// bounds staleness for out-of-band edits; explicit invalidation is unnecessary
|
||||
// because the key includes the effective media kind, theatrical flag, query
|
||||
// and year. Negative entries use a shorter TTL so a failed query is retried
|
||||
// sooner without manual intervention.
|
||||
const (
|
||||
scrapeLookupCacheTTL = 10 * time.Minute
|
||||
scrapeLookupNegativeCacheTTL = 2 * time.Minute
|
||||
scrapeLookupCacheMaxItems = 1024
|
||||
)
|
||||
|
||||
type scrapeLookupCache struct {
|
||||
mu sync.Mutex
|
||||
entries map[string]scrapeLookupCacheEntry
|
||||
}
|
||||
|
||||
type scrapeLookupCacheEntry struct {
|
||||
match *Match
|
||||
expiresAt time.Time
|
||||
}
|
||||
|
||||
func newScrapeLookupCache() *scrapeLookupCache {
|
||||
return &scrapeLookupCache{entries: map[string]scrapeLookupCacheEntry{}}
|
||||
}
|
||||
|
||||
func scrapeLookupCacheKey(kind, query string, year int, isTheatrical bool) string {
|
||||
theatrical := "0"
|
||||
if isTheatrical {
|
||||
theatrical = "1"
|
||||
}
|
||||
return strings.ToLower(strings.TrimSpace(kind)) + "|" +
|
||||
theatrical + "|" +
|
||||
strconv.Itoa(year) + "|" +
|
||||
strings.ToLower(strings.TrimSpace(query))
|
||||
}
|
||||
|
||||
// get reports whether the key is cached. A hit may carry a nil match, which
|
||||
// means the provider chain already ran and found nothing (negative cache).
|
||||
// Use clearNegative before a user-triggered retry so stale negative entries
|
||||
// do not suppress the fresh provider round-trip.
|
||||
func (c *scrapeLookupCache) get(key string) (*Match, bool) {
|
||||
if c == nil || key == "" {
|
||||
return nil, false
|
||||
}
|
||||
now := time.Now()
|
||||
c.mu.Lock()
|
||||
defer c.mu.Unlock()
|
||||
item, ok := c.entries[key]
|
||||
if !ok {
|
||||
return nil, false
|
||||
}
|
||||
if now.After(item.expiresAt) {
|
||||
delete(c.entries, key)
|
||||
return nil, false
|
||||
}
|
||||
return cloneMatch(item.match), true
|
||||
}
|
||||
|
||||
func (c *scrapeLookupCache) set(key string, match *Match) {
|
||||
if c == nil || key == "" {
|
||||
return
|
||||
}
|
||||
now := time.Now()
|
||||
c.mu.Lock()
|
||||
defer c.mu.Unlock()
|
||||
if len(c.entries) >= scrapeLookupCacheMaxItems {
|
||||
for k, item := range c.entries {
|
||||
if now.After(item.expiresAt) || len(c.entries) >= scrapeLookupCacheMaxItems {
|
||||
delete(c.entries, k)
|
||||
}
|
||||
if len(c.entries) < scrapeLookupCacheMaxItems {
|
||||
break
|
||||
}
|
||||
}
|
||||
}
|
||||
ttl := scrapeLookupCacheTTL
|
||||
if match == nil {
|
||||
ttl = scrapeLookupNegativeCacheTTL
|
||||
}
|
||||
c.entries[key] = scrapeLookupCacheEntry{match: cloneMatch(match), expiresAt: now.Add(ttl)}
|
||||
}
|
||||
|
||||
// clearNegatives drops cached misses so a user-triggered retry gets a fresh
|
||||
// provider round-trip instead of reusing a stale negative entry.
|
||||
func (c *scrapeLookupCache) clearNegatives() {
|
||||
if c == nil {
|
||||
return
|
||||
}
|
||||
c.mu.Lock()
|
||||
defer c.mu.Unlock()
|
||||
for k, item := range c.entries {
|
||||
if item.match == nil {
|
||||
delete(c.entries, k)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// cloneMatch returns an independent copy so callers that mutate the result
|
||||
// (localized title preference, local metadata merge, fanart artwork) cannot
|
||||
// corrupt the cached entry or leak state between media rows.
|
||||
func cloneMatch(m *Match) *Match {
|
||||
if m == nil {
|
||||
return nil
|
||||
}
|
||||
out := *m
|
||||
out.Languages = append([]string(nil), m.Languages...)
|
||||
out.Countries = append([]string(nil), m.Countries...)
|
||||
out.Genres = append([]string(nil), m.Genres...)
|
||||
out.Aliases = append([]string(nil), m.Aliases...)
|
||||
return &out
|
||||
}
|
||||
@@ -0,0 +1,165 @@
|
||||
package service
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"sync/atomic"
|
||||
"testing"
|
||||
|
||||
"github.com/glebarez/sqlite"
|
||||
"go.uber.org/zap"
|
||||
"gorm.io/gorm"
|
||||
|
||||
"github.com/truewhile/MeBox/internal/config"
|
||||
"github.com/truewhile/MeBox/internal/model"
|
||||
"github.com/truewhile/MeBox/internal/repository"
|
||||
)
|
||||
|
||||
// Every episode of a show produces the same query candidate (the series folder
|
||||
// title). Without the lookup cache each episode re-issues the identical search,
|
||||
// so a whole season costs one request per episode. This asserts the provider is
|
||||
// hit once and subsequent episodes reuse the cached match.
|
||||
func TestEnrichLibraryReusesLookupCacheAcrossEpisodes(t *testing.T) {
|
||||
var searchCalls int32
|
||||
upstream := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
w.Header().Set("Content-Type", "application/json")
|
||||
switch r.URL.Path {
|
||||
case "/search/tv":
|
||||
if r.URL.Query().Get("query") != "折腰" {
|
||||
_ = json.NewEncoder(w).Encode(map[string]any{"results": []any{}})
|
||||
return
|
||||
}
|
||||
atomic.AddInt32(&searchCalls, 1)
|
||||
_ = json.NewEncoder(w).Encode(map[string]any{
|
||||
"results": []map[string]any{{
|
||||
"id": 296753,
|
||||
"name": "折腰",
|
||||
"overview": "正确的剧集条目",
|
||||
"poster_path": "/zheyao.jpg",
|
||||
"first_air_date": "2025-05-13",
|
||||
"origin_country": []string{"CN"},
|
||||
}},
|
||||
})
|
||||
default:
|
||||
http.NotFound(w, r)
|
||||
}
|
||||
}))
|
||||
defer upstream.Close()
|
||||
|
||||
db, err := gorm.Open(sqlite.Open("file::memory:?cache=shared"), &gorm.Config{})
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := db.AutoMigrate(&model.Library{}, &model.Series{}, &model.Media{}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
repos := repository.New(db)
|
||||
cfg := &config.Config{}
|
||||
cfg.Secrets.TMDbAPIKey = "test-key"
|
||||
cfg.Secrets.TMDbAPIProxy = upstream.URL
|
||||
cfg.Secrets.TMDbImageProxy = upstream.URL + "/images"
|
||||
log := zap.NewNop()
|
||||
scraper := NewScraperService(cfg, log, repos, NewTMDbProvider(cfg, log, nil), nil, nil, nil, NewHub(log))
|
||||
|
||||
lib := model.Library{Name: "OpenList · 刮削缓存测试库", Path: "cloud://openlist/scrape-lookup-cache", Type: "tv", Enabled: true}
|
||||
if err := repos.DB.Create(&lib).Error; err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
const episodeCount = 4
|
||||
for episode := 1; episode <= episodeCount; episode++ {
|
||||
media := model.Media{
|
||||
LibraryID: lib.ID,
|
||||
Title: "折腰",
|
||||
Path: "cloud://openlist/scrape-lookup-cache/折腰 (2025)/Season 1/折腰.S01E0" + string(rune('0'+episode)) + ".mkv",
|
||||
SeasonNum: 1,
|
||||
EpisodeNum: episode,
|
||||
ScrapeStatus: "pending",
|
||||
}
|
||||
if err := repos.DB.Create(&media).Error; err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
|
||||
result, err := scraper.EnrichLibraryDetailedWithOptions(t.Context(), lib.ID, ScrapeOptions{})
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if result.Matched != episodeCount {
|
||||
t.Fatalf("matched = %d, want %d", result.Matched, episodeCount)
|
||||
}
|
||||
if calls := atomic.LoadInt32(&searchCalls); calls != 1 {
|
||||
t.Fatalf("tmdb /search/tv called %d times for %d episodes; want 1 (cache should dedupe identical queries)", calls, episodeCount)
|
||||
}
|
||||
}
|
||||
|
||||
// A cached match must be handed out as an independent copy: mutating one row's
|
||||
// result (localized title preference, local metadata merge) must not bleed into
|
||||
// the next row.
|
||||
func TestScrapeLookupCacheReturnsIndependentCopies(t *testing.T) {
|
||||
cache := newScrapeLookupCache()
|
||||
original := &Match{
|
||||
Title: "折腰",
|
||||
TMDbID: 296753,
|
||||
Genres: []string{"剧情"},
|
||||
Countries: []string{"CN"},
|
||||
Aliases: []string{"Zhe Yao"},
|
||||
}
|
||||
key := scrapeLookupCacheKey("tv", "折腰", 2025, false)
|
||||
cache.set(key, original)
|
||||
|
||||
first, ok := cache.get(key)
|
||||
if !ok || first == nil {
|
||||
t.Fatal("expected a cache hit")
|
||||
}
|
||||
first.Title = "mutated"
|
||||
first.Genres[0] = "mutated"
|
||||
first.Aliases = append(first.Aliases, "extra")
|
||||
|
||||
second, ok := cache.get(key)
|
||||
if !ok || second == nil {
|
||||
t.Fatal("expected a second cache hit")
|
||||
}
|
||||
if second.Title != "折腰" || second.Genres[0] != "剧情" || len(second.Aliases) != 1 {
|
||||
t.Fatalf("cached match was mutated through a returned copy: %+v", second)
|
||||
}
|
||||
}
|
||||
|
||||
// Negative results are cached too: a query that matched nothing must not be
|
||||
// re-issued for every remaining episode of the same show.
|
||||
func TestScrapeLookupCacheStoresNegativeResults(t *testing.T) {
|
||||
cache := newScrapeLookupCache()
|
||||
key := scrapeLookupCacheKey("tv", "no-such-show", 0, false)
|
||||
cache.set(key, nil)
|
||||
if _, ok := cache.get(key); !ok {
|
||||
t.Fatal("negative result should be cached to avoid repeated provider calls")
|
||||
}
|
||||
}
|
||||
|
||||
// The theatrical flag is part of the key: a theatrical feature gets a tv
|
||||
// fallback lookup, so its result must not be shared with (or returned for)
|
||||
// the same query issued for a non-theatrical media row.
|
||||
func TestScrapeLookupCacheKeySeparatesTheatrical(t *testing.T) {
|
||||
plain := scrapeLookupCacheKey("movie", "query", 2024, false)
|
||||
theatrical := scrapeLookupCacheKey("movie", "query", 2024, true)
|
||||
if plain == theatrical {
|
||||
t.Fatalf("theatrical flag must be part of the cache key: %q", plain)
|
||||
}
|
||||
}
|
||||
|
||||
// A manual "retry no match" run clears stale negative entries so the provider
|
||||
// chain is actually re-queried instead of short-circuiting on the cached miss.
|
||||
func TestScrapeLookupCacheClearNegativesKeepsPositives(t *testing.T) {
|
||||
cache := newScrapeLookupCache()
|
||||
negKey := scrapeLookupCacheKey("tv", "missing", 0, false)
|
||||
posKey := scrapeLookupCacheKey("tv", "found", 0, false)
|
||||
cache.set(negKey, nil)
|
||||
cache.set(posKey, &Match{Title: "found"})
|
||||
cache.clearNegatives()
|
||||
if _, ok := cache.get(negKey); ok {
|
||||
t.Fatal("negative entry should be cleared before a manual retry")
|
||||
}
|
||||
if got, ok := cache.get(posKey); !ok || got == nil || got.Title != "found" {
|
||||
t.Fatalf("positive entry should survive clearNegatives: %+v", got)
|
||||
}
|
||||
}
|
||||
@@ -24,11 +24,15 @@ var noiseTokens = []string{
|
||||
"hkfree", "yify", "rarbg", "ettv", "fgt", "tgx", "ctrlhd", "ntb", "flux", "qhstudio",
|
||||
|
||||
// 流媒体平台 / 字幕组 / 国家版本(动漫常见)
|
||||
"netflix", "nf", "amzn", "hulu", "disney", "max", "hbo",
|
||||
// 注意:不要把同时是常见英文单词的标记放进来(如 max / web / judas),
|
||||
// 否则 "Mad Max"、"Web Therapy" 这类正常标题会被误删。裸 "WEB" 发布标记
|
||||
// 通常位于分辨率之后,由 releaseBoundary 截断规则处理;"WEB-DL" 则由
|
||||
// multiWordNoise 单独匹配。
|
||||
"netflix", "nf", "amzn", "hulu", "disney", "hbo",
|
||||
"linetv", "ourtv", "iqiyi", "youku", "bilibili", "qiyi", "krj",
|
||||
"atvp", "appletv", "apple-tv", "tx", "txweb",
|
||||
"crunchyroll", "funimation", "anidb", "horriblesubs", "subsplease",
|
||||
"erai-raws", "judas", "asw", "smcat", "leopard-raws", "ohys-raws", "colortv",
|
||||
"erai-raws", "asw", "smcat", "leopard-raws", "ohys-raws", "colortv",
|
||||
"mweb", "ubweb", "hhweb", "adweb", "chdweb", "kurosawa", "qhstudio",
|
||||
|
||||
// 中文字幕标记
|
||||
@@ -55,6 +59,20 @@ var releaseBoundaryTokenSet = map[string]struct{}{
|
||||
"x264": {}, "x265": {}, "h264": {}, "h265": {}, "h266": {}, "hevc": {}, "avc": {}, "av1": {}, "vvc": {},
|
||||
}
|
||||
|
||||
// weakReleaseBoundaryTokenSet are release tags that double as plausible title
|
||||
// words. Before any release signal has been seen they are kept as part of the
|
||||
// title ("Mad Max", "Web Therapy"), so a title is never truncated to nothing;
|
||||
// after a real signal they behave like any other tag.
|
||||
var weakReleaseBoundaryTokenSet = map[string]struct{}{
|
||||
"bd": {}, "dvd": {}, "web": {},
|
||||
}
|
||||
|
||||
// releaseSignalToken marks the position of an extracted year. The year itself
|
||||
// is not a title token, but its presence still proves the following tokens are
|
||||
// release tags ("复仇者联盟4.2019.BD.1080p" must drop "BD"). A control rune is
|
||||
// used so it can never collide with a real filename token.
|
||||
const releaseSignalToken = "\u0001"
|
||||
|
||||
var dynamicReleaseBoundaryTokenRE = regexp.MustCompile(`(?i)^(?:\d{3,4}p|\d{2,3}fps)$`)
|
||||
|
||||
// bracketedTag matches "[anything]", "(anything)" or "{anything}" segments.
|
||||
@@ -86,7 +104,9 @@ func CleanQuery(raw string) (title string, year int) {
|
||||
if m := yearPattern.FindStringSubmatch(lower); len(m) >= 2 {
|
||||
if v, err := strconv.Atoi(m[1]); err == nil {
|
||||
year = v
|
||||
lower = strings.ReplaceAll(lower, m[1], " ")
|
||||
// Keep a positional marker: the year is not a title token, but its
|
||||
// presence arms release-tag truncation for what follows.
|
||||
lower = strings.ReplaceAll(lower, m[1], " "+releaseSignalToken+" ")
|
||||
}
|
||||
}
|
||||
|
||||
@@ -110,17 +130,35 @@ func CleanQuery(raw string) (title string, year int) {
|
||||
}
|
||||
// 拆分后丢掉过短(≤1)且全为 ASCII 数字 / 字母的"碎片",避免
|
||||
// 「2」「0」「v」之类残留干扰 TMDb 搜索。中文字符不算碎片。
|
||||
//
|
||||
// seenReleaseBoundary 只有在遇到分辨率/编码等强标记后才置位,置位后其余
|
||||
// ASCII 词一律视为发布尾巴丢弃;releaseSignalled 则宽松得多,年份标记也会
|
||||
// 置位,它只用来武装弱标记(bd/dvd/web)。这样 "Web Therapy" 不会被截空,
|
||||
// 而 "2019.Avatar.1080p" 这种年份前置的标题也不会因为年份标记把后面的
|
||||
// 真实标题词当成尾巴丢掉。
|
||||
out := make([]string, 0, 8)
|
||||
seenReleaseBoundary := false
|
||||
releaseSignalled := false
|
||||
for _, w := range strings.Fields(lower) {
|
||||
if w == releaseSignalToken {
|
||||
releaseSignalled = true
|
||||
continue
|
||||
}
|
||||
if dynamicReleaseBoundaryTokenRE.MatchString(w) {
|
||||
releaseSignalled = true
|
||||
seenReleaseBoundary = true
|
||||
continue
|
||||
}
|
||||
if _, ok := noiseTokenSet[w]; ok {
|
||||
if _, boundary := releaseBoundaryTokenSet[w]; boundary {
|
||||
if _, boundary := releaseBoundaryTokenSet[w]; boundary {
|
||||
if _, weak := weakReleaseBoundaryTokenSet[w]; weak && !releaseSignalled {
|
||||
// Ambiguous tag that is also a plausible title word; keep it
|
||||
// until a real release signal proves we are past the title.
|
||||
} else {
|
||||
releaseSignalled = true
|
||||
seenReleaseBoundary = true
|
||||
continue
|
||||
}
|
||||
} else if _, ok := noiseTokenSet[w]; ok {
|
||||
continue
|
||||
}
|
||||
if seenReleaseBoundary && isASCIIWord(w) {
|
||||
|
||||
@@ -0,0 +1,77 @@
|
||||
package service
|
||||
|
||||
import "testing"
|
||||
|
||||
// CleanQuery must not delete words that are also ordinary English title words
|
||||
// just because a release group reused them as a tag. Regression guard: "max"
|
||||
// and "web" used to be unconditional noise/boundary tokens, which turned
|
||||
// "Mad Max" into "mad" and "Web Therapy" into an empty query.
|
||||
func TestCleanQueryKeepsCommonEnglishTitleWords(t *testing.T) {
|
||||
cases := []struct {
|
||||
in string
|
||||
wantTitle string
|
||||
wantYear int
|
||||
}{
|
||||
{"Mad.Max.1979.1080p.BluRay.mkv", "mad max", 1979},
|
||||
{"Max.Payne.2008.1080p.WEB-DL.mkv", "max payne", 2008},
|
||||
{"Web.Therapy.S01E01.1080p.WEB-DL.mkv", "web therapy", 0},
|
||||
{"The.Web.2019.1080p.mkv", "the web", 2019},
|
||||
}
|
||||
for _, tc := range cases {
|
||||
t.Run(tc.in, func(t *testing.T) {
|
||||
gotTitle, gotYear := CleanQuery(tc.in)
|
||||
if gotTitle != tc.wantTitle || gotYear != tc.wantYear {
|
||||
t.Errorf("CleanQuery(%q) = (%q, %d), want (%q, %d)",
|
||||
tc.in, gotTitle, gotYear, tc.wantTitle, tc.wantYear)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// A title consisting only of an ambiguous tag plus a release tail must not be
|
||||
// truncated to nothing; the tag stays so the query still has a chance.
|
||||
func TestCleanQueryDoesNotTruncateAmbiguousTagToNothing(t *testing.T) {
|
||||
for _, in := range []string{"Web.1080p.WEB-DL.mkv", "BD.720p.mkv"} {
|
||||
title, _ := CleanQuery(in)
|
||||
if title == "" {
|
||||
t.Errorf("CleanQuery(%q) returned an empty title", in)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// A year-prefixed filename must keep its title: the year arms weak-tag
|
||||
// truncation but must not discard the real title words that follow it.
|
||||
func TestCleanQueryKeepsTitleAfterYearPrefix(t *testing.T) {
|
||||
cases := map[string]string{
|
||||
"2019.Avatar.1080p.BluRay.mkv": "avatar",
|
||||
"2024.Dune.Part.Two.2160p.WEB.mkv": "dune part two",
|
||||
}
|
||||
for in, want := range cases {
|
||||
t.Run(in, func(t *testing.T) {
|
||||
got, year := CleanQuery(in)
|
||||
if got != want {
|
||||
t.Errorf("CleanQuery(%q) = (%q, %d), want title %q", in, got, year, want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// Release-tag truncation must still fire once a real release signal (an
|
||||
// extracted year, resolution or codec) has been seen, so tags after it are
|
||||
// dropped as before.
|
||||
func TestCleanQueryStillDropsTagsAfterReleaseSignal(t *testing.T) {
|
||||
cases := map[string]string{
|
||||
"复仇者联盟4.2019.BD.1080p.mkv": "复仇者联盟4",
|
||||
"The.Matrix.1999.1080p.WEB-DL.H265.mp4": "the matrix",
|
||||
"Oppenheimer.2023.2160p.UHD.BluRay.mkv": "oppenheimer",
|
||||
"Interstellar.2014.4k.hdr.dts.atmos.mkv": "interstellar",
|
||||
}
|
||||
for in, want := range cases {
|
||||
t.Run(in, func(t *testing.T) {
|
||||
got, _ := CleanQuery(in)
|
||||
if got != want {
|
||||
t.Errorf("CleanQuery(%q) = %q, want %q", in, got, want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
@@ -26,6 +26,10 @@ type ScraperService struct {
|
||||
cache *RuntimeCacheService
|
||||
images *ImageProxy
|
||||
|
||||
// Caches provider-chain results keyed by media kind/query/year. Per-instance
|
||||
// so services with different provider config never share results.
|
||||
lookupCache *scrapeLookupCache
|
||||
|
||||
// Serializes final sidecar replacement. Windows cannot rename over an
|
||||
// existing file, and concurrent scrapes can target the same sidecar.
|
||||
artworkWriteMu sync.Mutex
|
||||
@@ -50,6 +54,7 @@ func NewScraperService(
|
||||
return &ScraperService{
|
||||
cfg: cfg, log: log, repo: repo,
|
||||
tmdb: tmdb, bangumi: bangumi, thetvdb: thetvdb, fanart: fanart, adult: adultProvider, hub: hub,
|
||||
lookupCache: newScrapeLookupCache(),
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
Reference in New Issue
Block a user