This commit is contained in:
truewhile
2026-09-10 16:58:41 +08:00
parent 8f2551a1b6
commit b855e00345
11 changed files with 632 additions and 35 deletions
@@ -55,7 +55,7 @@ func (o *OrganizerService) lookupOrganizeMetadata(ctx context.Context, src, sour
continue
}
}
match := o.scraper.lookup(ctx, lib, media, candidate, year)
match := o.scraper.lookup(ctx, lib, media, candidate, year, false)
if match != nil && strings.TrimSpace(match.Title) != "" {
if !organizeMetadataMatchTrusted(candidate, year, match) {
if cache != nil {
+73 -6
View File
@@ -5,8 +5,10 @@ import (
"encoding/json"
"fmt"
"net/http"
"regexp"
"strconv"
"strings"
"sync"
"time"
"go.uber.org/zap"
@@ -61,8 +63,9 @@ func NewRecognitionWordsService(log *zap.Logger, repo *repository.Container) *Re
}
func (s *RecognitionWordsService) Config(ctx context.Context) RecognitionWordsConfig {
cfg := recognitionWordsConfig(ctx, s.repo)
cfg.RuleCount = len(parseRecognitionWordRules(recognitionWordsCombinedText(cfg)))
cfg, rules := recognitionWordsRuntime(ctx, s.repo)
cfg = cloneRecognitionWordsConfig(cfg)
cfg.RuleCount = len(rules)
return cfg
}
@@ -80,7 +83,11 @@ func (s *RecognitionWordsService) SaveConfig(ctx context.Context, cfg Recognitio
if err != nil {
return err
}
return s.repo.Setting.Set(ctx, RecognitionWordsSharedURLsKey, string(rawURLs))
if err := s.repo.Setting.Set(ctx, RecognitionWordsSharedURLsKey, string(rawURLs)); err != nil {
return err
}
invalidateRecognitionWordsRuntime(s.repo)
return nil
}
func (s *RecognitionWordsService) SyncShared(ctx context.Context) (RecognitionWordsConfig, error) {
@@ -104,6 +111,7 @@ func (s *RecognitionWordsService) SyncShared(ctx context.Context) (RecognitionWo
if err := s.repo.Setting.Set(ctx, RecognitionWordsSyncedAtKey, now); err != nil {
return cfg, err
}
invalidateRecognitionWordsRuntime(s.repo)
return s.Config(ctx), nil
}
@@ -120,11 +128,10 @@ func (s *RecognitionWordsService) Test(ctx context.Context, input string) Recogn
}
func ApplyRecognitionWords(ctx context.Context, repo *repository.Container, raw string) string {
cfg := recognitionWordsConfig(ctx, repo)
if !cfg.Enabled {
cfg, rules := recognitionWordsRuntime(ctx, repo)
if !cfg.Enabled || len(rules) == 0 {
return raw
}
rules := parseRecognitionWordRules(recognitionWordsCombinedText(cfg))
return applyRecognitionWordRules(raw, rules)
}
@@ -189,9 +196,69 @@ func normalizeRecognitionWordURLs(values []string) []string {
type recognitionWordRule struct {
raw string
block string
blockRE *regexp.Regexp
replaceFrom string
replaceTo string
replaceRE *regexp.Regexp
offsetLeft string
offsetRight string
offsetExpr string
offsetRE *regexp.Regexp
}
// recognitionWordsRuntime caches the parsed rule set (with pre-compiled
// regexes) per repository. Reading five settings rows and parsing/recompiling
// the whole shared word list used to happen on every query candidate, which is
// the scraper's hottest path. Writes through SaveConfig/SyncShared invalidate
// the entry immediately; the TTL only bounds staleness for out-of-process edits.
type recognitionWordsRuntimeEntry struct {
repo *repository.Container
cfg RecognitionWordsConfig
rules []recognitionWordRule
expiresAt time.Time
}
const recognitionWordsRuntimeTTL = 30 * time.Second
var recognitionWordsRuntimeCache struct {
sync.RWMutex
entry recognitionWordsRuntimeEntry
valid bool
}
func recognitionWordsRuntime(ctx context.Context, repo *repository.Container) (RecognitionWordsConfig, []recognitionWordRule) {
now := time.Now()
recognitionWordsRuntimeCache.RLock()
cached := recognitionWordsRuntimeCache.entry
hit := recognitionWordsRuntimeCache.valid && cached.repo == repo && now.Before(cached.expiresAt)
recognitionWordsRuntimeCache.RUnlock()
if hit {
return cached.cfg, cached.rules
}
cfg := recognitionWordsConfig(ctx, repo)
rules := parseRecognitionWordRules(recognitionWordsCombinedText(cfg))
recognitionWordsRuntimeCache.Lock()
recognitionWordsRuntimeCache.entry = recognitionWordsRuntimeEntry{
repo: repo,
cfg: cfg,
rules: rules,
expiresAt: now.Add(recognitionWordsRuntimeTTL),
}
recognitionWordsRuntimeCache.valid = true
recognitionWordsRuntimeCache.Unlock()
return cfg, rules
}
func invalidateRecognitionWordsRuntime(repo *repository.Container) {
recognitionWordsRuntimeCache.Lock()
if recognitionWordsRuntimeCache.entry.repo == repo {
recognitionWordsRuntimeCache.valid = false
}
recognitionWordsRuntimeCache.Unlock()
}
func cloneRecognitionWordsConfig(cfg RecognitionWordsConfig) RecognitionWordsConfig {
cfg.SharedURLs = append([]string(nil), cfg.SharedURLs...)
return cfg
}
@@ -0,0 +1,49 @@
package service
import (
"testing"
"go.uber.org/zap"
)
// The rule set is cached process-wide, so a config write must invalidate it
// immediately; otherwise the admin would have to wait out the TTL before a
// saved word list takes effect.
func TestRecognitionWordsCacheInvalidatedBySaveConfig(t *testing.T) {
repos := newOrganizerTestRepo(t)
svc := NewRecognitionWordsService(zap.NewNop(), repos)
if err := svc.SaveConfig(t.Context(), RecognitionWordsConfig{
Enabled: true,
LocalText: "BADWORD => 好标题",
}); err != nil {
t.Fatal(err)
}
if got := ApplyRecognitionWords(t.Context(), repos, "BADWORD"); got != "好标题" {
t.Fatalf("first apply = %q, want 好标题", got)
}
if err := svc.SaveConfig(t.Context(), RecognitionWordsConfig{
Enabled: true,
LocalText: "BADWORD => 新标题",
}); err != nil {
t.Fatal(err)
}
if got := ApplyRecognitionWords(t.Context(), repos, "BADWORD"); got != "新标题" {
t.Fatalf("cached rules were not invalidated after SaveConfig: got %q, want 新标题", got)
}
}
func TestRecognitionWordsDisabledReturnsRawInput(t *testing.T) {
repos := newOrganizerTestRepo(t)
svc := NewRecognitionWordsService(zap.NewNop(), repos)
if err := svc.SaveConfig(t.Context(), RecognitionWordsConfig{
Enabled: false,
LocalText: "BADWORD => 好标题",
}); err != nil {
t.Fatal(err)
}
if got := ApplyRecognitionWords(t.Context(), repos, "BADWORD"); got != "BADWORD" {
t.Fatalf("disabled recognition words changed input: got %q", got)
}
}
+59 -21
View File
@@ -31,6 +31,11 @@ func parseRecognitionWordRule(line string) recognitionWordRule {
case strings.Contains(part, "<>") && strings.Contains(part, ">>"):
beforeAfter := strings.SplitN(part, ">>", 2)
bounds := strings.SplitN(beforeAfter[0], "<>", 2)
// A malformed rule without "<>" would otherwise index past the
// slice; word lists are fetched from the network, so stay defensive.
if len(bounds) < 2 {
continue
}
rule.offsetLeft = strings.TrimSpace(bounds[0])
rule.offsetRight = strings.TrimSpace(bounds[1])
rule.offsetExpr = strings.TrimSpace(beforeAfter[1])
@@ -38,56 +43,89 @@ func parseRecognitionWordRule(line string) recognitionWordRule {
rule.block = part
}
}
compileRecognitionWordRule(&rule)
return rule
}
var recognitionReplacementRE = regexp.MustCompile(`\\([0-9]+)`)
func normalizeRecognitionReplacement(value string) string {
re := regexp.MustCompile(`\\([0-9]+)`)
return re.ReplaceAllString(value, "$$$1")
return recognitionReplacementRE.ReplaceAllString(value, "$$$1")
}
// compileRecognitionWordRule pre-compiles every pattern in a rule so the hot
// clean-query path never recompiles regexes per candidate.
func compileRecognitionWordRule(rule *recognitionWordRule) {
if rule == nil {
return
}
if rule.block != "" {
if re, err := regexp.Compile(rule.block); err == nil {
rule.blockRE = re
}
}
if rule.replaceFrom != "" {
if re, err := regexp.Compile(rule.replaceFrom); err == nil {
rule.replaceRE = re
}
}
if rule.offsetExpr != "" && (rule.offsetLeft != "" || rule.offsetRight != "") {
if re, err := compileRecognitionOffsetRE(rule.offsetLeft, rule.offsetRight); err == nil {
rule.offsetRE = re
}
}
}
func compileRecognitionOffsetRE(left, right string) (*regexp.Regexp, error) {
leftPattern := firstNonEmpty(left, `^`)
rightPattern := firstNonEmpty(right, `$`)
return regexp.Compile(`(?i)(` + leftPattern + `)(\d{1,5})(` + rightPattern + `)`)
}
func applyRecognitionWordRules(raw string, rules []recognitionWordRule) string {
out := strings.TrimSpace(raw)
for _, rule := range rules {
if rule.block != "" {
out = applyRecognitionBlock(out, rule.block)
out = applyRecognitionBlock(out, rule)
}
if rule.replaceFrom != "" {
out = applyRecognitionReplace(out, rule.replaceFrom, rule.replaceTo)
out = applyRecognitionReplace(out, rule)
}
if rule.offsetLeft != "" || rule.offsetRight != "" {
out = applyRecognitionOffset(out, rule.offsetLeft, rule.offsetRight, rule.offsetExpr)
out = applyRecognitionOffset(out, rule)
}
}
return strings.Join(strings.Fields(out), " ")
}
func applyRecognitionBlock(raw, block string) string {
if re, err := regexp.Compile(block); err == nil {
return re.ReplaceAllString(raw, " ")
func applyRecognitionBlock(raw string, rule recognitionWordRule) string {
if rule.blockRE != nil {
return rule.blockRE.ReplaceAllString(raw, " ")
}
return strings.ReplaceAll(raw, block, " ")
return strings.ReplaceAll(raw, rule.block, " ")
}
func applyRecognitionReplace(raw, from, to string) string {
if re, err := regexp.Compile(from); err == nil {
return re.ReplaceAllString(raw, to)
func applyRecognitionReplace(raw string, rule recognitionWordRule) string {
if rule.replaceRE != nil {
return rule.replaceRE.ReplaceAllString(raw, rule.replaceTo)
}
return strings.ReplaceAll(raw, from, to)
return strings.ReplaceAll(raw, rule.replaceFrom, rule.replaceTo)
}
func applyRecognitionOffset(raw, left, right, expr string) string {
if strings.TrimSpace(expr) == "" {
func applyRecognitionOffset(raw string, rule recognitionWordRule) string {
if strings.TrimSpace(rule.offsetExpr) == "" {
return raw
}
leftPattern := firstNonEmpty(left, `^`)
rightPattern := firstNonEmpty(right, `$`)
re, err := regexp.Compile(`(?i)(` + leftPattern + `)(\d{1,5})(` + rightPattern + `)`)
if err != nil {
return raw
re := rule.offsetRE
if re == nil {
compiled, err := compileRecognitionOffsetRE(rule.offsetLeft, rule.offsetRight)
if err != nil {
return raw
}
re = compiled
}
return re.ReplaceAllStringFunc(raw, func(match string) string {
return applyRecognitionOffsetMatch(re, match, expr)
return applyRecognitionOffsetMatch(re, match, rule.offsetExpr)
})
}
+6 -1
View File
@@ -63,12 +63,17 @@ func (s *ScraperService) EnrichOneWithOptions(ctx context.Context, m *model.Medi
}
}
// Same stale-negative problem as the library path: a single-item retry
// (queue task / manual rescrape) must get a fresh provider round-trip.
if options.RetryNoMatch && s != nil {
s.lookupCache.clearNegatives()
}
candidates := scrapeQueryCandidatesWithRecognition(ctx, s.repo, m, lib)
var query string
match := (*Match)(nil)
for _, candidate := range candidates {
query = candidate
candidateMatch := s.lookup(ctx, lib, m, candidate, year)
candidateMatch := s.lookup(ctx, lib, m, candidate, year, options.ForceRematch)
if candidateMatch == nil {
continue
}
+28 -1
View File
@@ -13,7 +13,12 @@ import (
// lookup runs the provider chain after local NFO has been considered:
// TMDb -> Douban -> Bangumi -> TheTVDB. Douban and Bangumi do not require API
// keys; providers that are unavailable or return an error are skipped.
func (s *ScraperService) lookup(ctx context.Context, lib *model.Library, media *model.Media, query string, year int) *Match {
//
// Results are cached per (kind, theatrical, query, year) because every
// episode of a show produces the same candidate; bypassCache forces a fresh
// provider round-trip for user-triggered rematches but still refreshes the
// cache so later episodes reuse the corrected result.
func (s *ScraperService) lookup(ctx context.Context, lib *model.Library, media *model.Media, query string, year int, bypassCache bool) *Match {
kind := ""
if lib != nil {
kind = lib.Type
@@ -28,6 +33,21 @@ func (s *ScraperService) lookup(ctx context.Context, lib *model.Library, media *
} else if (normalizeOrganizeMediaType(kind) != "movie" || explicitEpisode) && mediaIsEpisodic(media, lib) {
kind = "tv"
}
cacheKey := scrapeLookupCacheKey(kind, query, year, isTheatrical)
if !bypassCache {
if cached, ok := s.lookupCache.get(cacheKey); ok {
return cached
}
}
match := s.lookupUncached(ctx, kind, isTheatrical, query, year)
// Always refresh the cache, even on bypassed (forced) lookups, so a
// corrected rematch replaces the stale entry for later episodes.
s.lookupCache.set(cacheKey, match)
return match
}
func (s *ScraperService) lookupUncached(ctx context.Context, kind string, isTheatrical bool, query string, year int) *Match {
if s.tmdb != nil && s.tmdb.Enabled() {
if match := s.lookupAutomaticTMDb(ctx, kind, query, year); match != nil {
match.Provider = "tmdb"
@@ -107,6 +127,13 @@ func (s *ScraperService) EnrichLibraryDetailed(ctx context.Context, libraryID st
func (s *ScraperService) EnrichLibraryDetailedWithOptions(ctx context.Context, libraryID string, options ScrapeOptions) (EnrichLibraryResult, error) {
result := EnrichLibraryResult{LibraryID: libraryID}
// A manual retry must not reuse stale negative cache entries from the
// previous run, otherwise "重新刮削" looks like it did nothing. Positives
// are kept so the rest of the season still dedupes; the first re-queried
// episode refreshes the entry for the rows behind it.
if options.RetryNoMatch && s != nil {
s.lookupCache.clearNegatives()
}
rows, err := s.scrapeCandidateRows(ctx, libraryID, options)
if err != nil {
return result, err
+126
View File
@@ -0,0 +1,126 @@
package service
import (
"strconv"
"strings"
"sync"
"time"
)
// A TV library scrapes one media row per episode, and every row in the same
// show computes the same query candidate (the series folder title). Without a
// cache each episode re-issues the identical TMDb/Douban/Bangumi/TheTVDB
// search, so a 100-episode season costs 100x the network calls it needs.
//
// The cache lives on the ScraperService instance (not a package global) so that
// two services with different provider configuration never share results. TTL
// bounds staleness for out-of-band edits; explicit invalidation is unnecessary
// because the key includes the effective media kind, theatrical flag, query
// and year. Negative entries use a shorter TTL so a failed query is retried
// sooner without manual intervention.
const (
scrapeLookupCacheTTL = 10 * time.Minute
scrapeLookupNegativeCacheTTL = 2 * time.Minute
scrapeLookupCacheMaxItems = 1024
)
type scrapeLookupCache struct {
mu sync.Mutex
entries map[string]scrapeLookupCacheEntry
}
type scrapeLookupCacheEntry struct {
match *Match
expiresAt time.Time
}
func newScrapeLookupCache() *scrapeLookupCache {
return &scrapeLookupCache{entries: map[string]scrapeLookupCacheEntry{}}
}
func scrapeLookupCacheKey(kind, query string, year int, isTheatrical bool) string {
theatrical := "0"
if isTheatrical {
theatrical = "1"
}
return strings.ToLower(strings.TrimSpace(kind)) + "|" +
theatrical + "|" +
strconv.Itoa(year) + "|" +
strings.ToLower(strings.TrimSpace(query))
}
// get reports whether the key is cached. A hit may carry a nil match, which
// means the provider chain already ran and found nothing (negative cache).
// Use clearNegative before a user-triggered retry so stale negative entries
// do not suppress the fresh provider round-trip.
func (c *scrapeLookupCache) get(key string) (*Match, bool) {
if c == nil || key == "" {
return nil, false
}
now := time.Now()
c.mu.Lock()
defer c.mu.Unlock()
item, ok := c.entries[key]
if !ok {
return nil, false
}
if now.After(item.expiresAt) {
delete(c.entries, key)
return nil, false
}
return cloneMatch(item.match), true
}
func (c *scrapeLookupCache) set(key string, match *Match) {
if c == nil || key == "" {
return
}
now := time.Now()
c.mu.Lock()
defer c.mu.Unlock()
if len(c.entries) >= scrapeLookupCacheMaxItems {
for k, item := range c.entries {
if now.After(item.expiresAt) || len(c.entries) >= scrapeLookupCacheMaxItems {
delete(c.entries, k)
}
if len(c.entries) < scrapeLookupCacheMaxItems {
break
}
}
}
ttl := scrapeLookupCacheTTL
if match == nil {
ttl = scrapeLookupNegativeCacheTTL
}
c.entries[key] = scrapeLookupCacheEntry{match: cloneMatch(match), expiresAt: now.Add(ttl)}
}
// clearNegatives drops cached misses so a user-triggered retry gets a fresh
// provider round-trip instead of reusing a stale negative entry.
func (c *scrapeLookupCache) clearNegatives() {
if c == nil {
return
}
c.mu.Lock()
defer c.mu.Unlock()
for k, item := range c.entries {
if item.match == nil {
delete(c.entries, k)
}
}
}
// cloneMatch returns an independent copy so callers that mutate the result
// (localized title preference, local metadata merge, fanart artwork) cannot
// corrupt the cached entry or leak state between media rows.
func cloneMatch(m *Match) *Match {
if m == nil {
return nil
}
out := *m
out.Languages = append([]string(nil), m.Languages...)
out.Countries = append([]string(nil), m.Countries...)
out.Genres = append([]string(nil), m.Genres...)
out.Aliases = append([]string(nil), m.Aliases...)
return &out
}
@@ -0,0 +1,165 @@
package service
import (
"encoding/json"
"net/http"
"net/http/httptest"
"sync/atomic"
"testing"
"github.com/glebarez/sqlite"
"go.uber.org/zap"
"gorm.io/gorm"
"github.com/truewhile/MeBox/internal/config"
"github.com/truewhile/MeBox/internal/model"
"github.com/truewhile/MeBox/internal/repository"
)
// Every episode of a show produces the same query candidate (the series folder
// title). Without the lookup cache each episode re-issues the identical search,
// so a whole season costs one request per episode. This asserts the provider is
// hit once and subsequent episodes reuse the cached match.
func TestEnrichLibraryReusesLookupCacheAcrossEpisodes(t *testing.T) {
var searchCalls int32
upstream := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
w.Header().Set("Content-Type", "application/json")
switch r.URL.Path {
case "/search/tv":
if r.URL.Query().Get("query") != "折腰" {
_ = json.NewEncoder(w).Encode(map[string]any{"results": []any{}})
return
}
atomic.AddInt32(&searchCalls, 1)
_ = json.NewEncoder(w).Encode(map[string]any{
"results": []map[string]any{{
"id": 296753,
"name": "折腰",
"overview": "正确的剧集条目",
"poster_path": "/zheyao.jpg",
"first_air_date": "2025-05-13",
"origin_country": []string{"CN"},
}},
})
default:
http.NotFound(w, r)
}
}))
defer upstream.Close()
db, err := gorm.Open(sqlite.Open("file::memory:?cache=shared"), &gorm.Config{})
if err != nil {
t.Fatal(err)
}
if err := db.AutoMigrate(&model.Library{}, &model.Series{}, &model.Media{}); err != nil {
t.Fatal(err)
}
repos := repository.New(db)
cfg := &config.Config{}
cfg.Secrets.TMDbAPIKey = "test-key"
cfg.Secrets.TMDbAPIProxy = upstream.URL
cfg.Secrets.TMDbImageProxy = upstream.URL + "/images"
log := zap.NewNop()
scraper := NewScraperService(cfg, log, repos, NewTMDbProvider(cfg, log, nil), nil, nil, nil, NewHub(log))
lib := model.Library{Name: "OpenList · 刮削缓存测试库", Path: "cloud://openlist/scrape-lookup-cache", Type: "tv", Enabled: true}
if err := repos.DB.Create(&lib).Error; err != nil {
t.Fatal(err)
}
const episodeCount = 4
for episode := 1; episode <= episodeCount; episode++ {
media := model.Media{
LibraryID: lib.ID,
Title: "折腰",
Path: "cloud://openlist/scrape-lookup-cache/折腰 (2025)/Season 1/折腰.S01E0" + string(rune('0'+episode)) + ".mkv",
SeasonNum: 1,
EpisodeNum: episode,
ScrapeStatus: "pending",
}
if err := repos.DB.Create(&media).Error; err != nil {
t.Fatal(err)
}
}
result, err := scraper.EnrichLibraryDetailedWithOptions(t.Context(), lib.ID, ScrapeOptions{})
if err != nil {
t.Fatal(err)
}
if result.Matched != episodeCount {
t.Fatalf("matched = %d, want %d", result.Matched, episodeCount)
}
if calls := atomic.LoadInt32(&searchCalls); calls != 1 {
t.Fatalf("tmdb /search/tv called %d times for %d episodes; want 1 (cache should dedupe identical queries)", calls, episodeCount)
}
}
// A cached match must be handed out as an independent copy: mutating one row's
// result (localized title preference, local metadata merge) must not bleed into
// the next row.
func TestScrapeLookupCacheReturnsIndependentCopies(t *testing.T) {
cache := newScrapeLookupCache()
original := &Match{
Title: "折腰",
TMDbID: 296753,
Genres: []string{"剧情"},
Countries: []string{"CN"},
Aliases: []string{"Zhe Yao"},
}
key := scrapeLookupCacheKey("tv", "折腰", 2025, false)
cache.set(key, original)
first, ok := cache.get(key)
if !ok || first == nil {
t.Fatal("expected a cache hit")
}
first.Title = "mutated"
first.Genres[0] = "mutated"
first.Aliases = append(first.Aliases, "extra")
second, ok := cache.get(key)
if !ok || second == nil {
t.Fatal("expected a second cache hit")
}
if second.Title != "折腰" || second.Genres[0] != "剧情" || len(second.Aliases) != 1 {
t.Fatalf("cached match was mutated through a returned copy: %+v", second)
}
}
// Negative results are cached too: a query that matched nothing must not be
// re-issued for every remaining episode of the same show.
func TestScrapeLookupCacheStoresNegativeResults(t *testing.T) {
cache := newScrapeLookupCache()
key := scrapeLookupCacheKey("tv", "no-such-show", 0, false)
cache.set(key, nil)
if _, ok := cache.get(key); !ok {
t.Fatal("negative result should be cached to avoid repeated provider calls")
}
}
// The theatrical flag is part of the key: a theatrical feature gets a tv
// fallback lookup, so its result must not be shared with (or returned for)
// the same query issued for a non-theatrical media row.
func TestScrapeLookupCacheKeySeparatesTheatrical(t *testing.T) {
plain := scrapeLookupCacheKey("movie", "query", 2024, false)
theatrical := scrapeLookupCacheKey("movie", "query", 2024, true)
if plain == theatrical {
t.Fatalf("theatrical flag must be part of the cache key: %q", plain)
}
}
// A manual "retry no match" run clears stale negative entries so the provider
// chain is actually re-queried instead of short-circuiting on the cached miss.
func TestScrapeLookupCacheClearNegativesKeepsPositives(t *testing.T) {
cache := newScrapeLookupCache()
negKey := scrapeLookupCacheKey("tv", "missing", 0, false)
posKey := scrapeLookupCacheKey("tv", "found", 0, false)
cache.set(negKey, nil)
cache.set(posKey, &Match{Title: "found"})
cache.clearNegatives()
if _, ok := cache.get(negKey); ok {
t.Fatal("negative entry should be cleared before a manual retry")
}
if got, ok := cache.get(posKey); !ok || got == nil || got.Title != "found" {
t.Fatalf("positive entry should survive clearNegatives: %+v", got)
}
}
+43 -5
View File
@@ -24,11 +24,15 @@ var noiseTokens = []string{
"hkfree", "yify", "rarbg", "ettv", "fgt", "tgx", "ctrlhd", "ntb", "flux", "qhstudio",
// 流媒体平台 / 字幕组 / 国家版本(动漫常见)
"netflix", "nf", "amzn", "hulu", "disney", "max", "hbo",
// 注意:不要把同时是常见英文单词的标记放进来(如 max / web / judas),
// 否则 "Mad Max"、"Web Therapy" 这类正常标题会被误删。裸 "WEB" 发布标记
// 通常位于分辨率之后,由 releaseBoundary 截断规则处理;"WEB-DL" 则由
// multiWordNoise 单独匹配。
"netflix", "nf", "amzn", "hulu", "disney", "hbo",
"linetv", "ourtv", "iqiyi", "youku", "bilibili", "qiyi", "krj",
"atvp", "appletv", "apple-tv", "tx", "txweb",
"crunchyroll", "funimation", "anidb", "horriblesubs", "subsplease",
"erai-raws", "judas", "asw", "smcat", "leopard-raws", "ohys-raws", "colortv",
"erai-raws", "asw", "smcat", "leopard-raws", "ohys-raws", "colortv",
"mweb", "ubweb", "hhweb", "adweb", "chdweb", "kurosawa", "qhstudio",
// 中文字幕标记
@@ -55,6 +59,20 @@ var releaseBoundaryTokenSet = map[string]struct{}{
"x264": {}, "x265": {}, "h264": {}, "h265": {}, "h266": {}, "hevc": {}, "avc": {}, "av1": {}, "vvc": {},
}
// weakReleaseBoundaryTokenSet are release tags that double as plausible title
// words. Before any release signal has been seen they are kept as part of the
// title ("Mad Max", "Web Therapy"), so a title is never truncated to nothing;
// after a real signal they behave like any other tag.
var weakReleaseBoundaryTokenSet = map[string]struct{}{
"bd": {}, "dvd": {}, "web": {},
}
// releaseSignalToken marks the position of an extracted year. The year itself
// is not a title token, but its presence still proves the following tokens are
// release tags ("复仇者联盟4.2019.BD.1080p" must drop "BD"). A control rune is
// used so it can never collide with a real filename token.
const releaseSignalToken = "\u0001"
var dynamicReleaseBoundaryTokenRE = regexp.MustCompile(`(?i)^(?:\d{3,4}p|\d{2,3}fps)$`)
// bracketedTag matches "[anything]", "(anything)" or "{anything}" segments.
@@ -86,7 +104,9 @@ func CleanQuery(raw string) (title string, year int) {
if m := yearPattern.FindStringSubmatch(lower); len(m) >= 2 {
if v, err := strconv.Atoi(m[1]); err == nil {
year = v
lower = strings.ReplaceAll(lower, m[1], " ")
// Keep a positional marker: the year is not a title token, but its
// presence arms release-tag truncation for what follows.
lower = strings.ReplaceAll(lower, m[1], " "+releaseSignalToken+" ")
}
}
@@ -110,17 +130,35 @@ func CleanQuery(raw string) (title string, year int) {
}
// 拆分后丢掉过短(≤1)且全为 ASCII 数字 / 字母的"碎片",避免
// 「2」「0」「v」之类残留干扰 TMDb 搜索。中文字符不算碎片。
//
// seenReleaseBoundary 只有在遇到分辨率/编码等强标记后才置位,置位后其余
// ASCII 词一律视为发布尾巴丢弃;releaseSignalled 则宽松得多,年份标记也会
// 置位,它只用来武装弱标记(bd/dvd/web)。这样 "Web Therapy" 不会被截空,
// 而 "2019.Avatar.1080p" 这种年份前置的标题也不会因为年份标记把后面的
// 真实标题词当成尾巴丢掉。
out := make([]string, 0, 8)
seenReleaseBoundary := false
releaseSignalled := false
for _, w := range strings.Fields(lower) {
if w == releaseSignalToken {
releaseSignalled = true
continue
}
if dynamicReleaseBoundaryTokenRE.MatchString(w) {
releaseSignalled = true
seenReleaseBoundary = true
continue
}
if _, ok := noiseTokenSet[w]; ok {
if _, boundary := releaseBoundaryTokenSet[w]; boundary {
if _, boundary := releaseBoundaryTokenSet[w]; boundary {
if _, weak := weakReleaseBoundaryTokenSet[w]; weak && !releaseSignalled {
// Ambiguous tag that is also a plausible title word; keep it
// until a real release signal proves we are past the title.
} else {
releaseSignalled = true
seenReleaseBoundary = true
continue
}
} else if _, ok := noiseTokenSet[w]; ok {
continue
}
if seenReleaseBoundary && isASCIIWord(w) {
@@ -0,0 +1,77 @@
package service
import "testing"
// CleanQuery must not delete words that are also ordinary English title words
// just because a release group reused them as a tag. Regression guard: "max"
// and "web" used to be unconditional noise/boundary tokens, which turned
// "Mad Max" into "mad" and "Web Therapy" into an empty query.
func TestCleanQueryKeepsCommonEnglishTitleWords(t *testing.T) {
cases := []struct {
in string
wantTitle string
wantYear int
}{
{"Mad.Max.1979.1080p.BluRay.mkv", "mad max", 1979},
{"Max.Payne.2008.1080p.WEB-DL.mkv", "max payne", 2008},
{"Web.Therapy.S01E01.1080p.WEB-DL.mkv", "web therapy", 0},
{"The.Web.2019.1080p.mkv", "the web", 2019},
}
for _, tc := range cases {
t.Run(tc.in, func(t *testing.T) {
gotTitle, gotYear := CleanQuery(tc.in)
if gotTitle != tc.wantTitle || gotYear != tc.wantYear {
t.Errorf("CleanQuery(%q) = (%q, %d), want (%q, %d)",
tc.in, gotTitle, gotYear, tc.wantTitle, tc.wantYear)
}
})
}
}
// A title consisting only of an ambiguous tag plus a release tail must not be
// truncated to nothing; the tag stays so the query still has a chance.
func TestCleanQueryDoesNotTruncateAmbiguousTagToNothing(t *testing.T) {
for _, in := range []string{"Web.1080p.WEB-DL.mkv", "BD.720p.mkv"} {
title, _ := CleanQuery(in)
if title == "" {
t.Errorf("CleanQuery(%q) returned an empty title", in)
}
}
}
// A year-prefixed filename must keep its title: the year arms weak-tag
// truncation but must not discard the real title words that follow it.
func TestCleanQueryKeepsTitleAfterYearPrefix(t *testing.T) {
cases := map[string]string{
"2019.Avatar.1080p.BluRay.mkv": "avatar",
"2024.Dune.Part.Two.2160p.WEB.mkv": "dune part two",
}
for in, want := range cases {
t.Run(in, func(t *testing.T) {
got, year := CleanQuery(in)
if got != want {
t.Errorf("CleanQuery(%q) = (%q, %d), want title %q", in, got, year, want)
}
})
}
}
// Release-tag truncation must still fire once a real release signal (an
// extracted year, resolution or codec) has been seen, so tags after it are
// dropped as before.
func TestCleanQueryStillDropsTagsAfterReleaseSignal(t *testing.T) {
cases := map[string]string{
"复仇者联盟4.2019.BD.1080p.mkv": "复仇者联盟4",
"The.Matrix.1999.1080p.WEB-DL.H265.mp4": "the matrix",
"Oppenheimer.2023.2160p.UHD.BluRay.mkv": "oppenheimer",
"Interstellar.2014.4k.hdr.dts.atmos.mkv": "interstellar",
}
for in, want := range cases {
t.Run(in, func(t *testing.T) {
got, _ := CleanQuery(in)
if got != want {
t.Errorf("CleanQuery(%q) = %q, want %q", in, got, want)
}
})
}
}
+5
View File
@@ -26,6 +26,10 @@ type ScraperService struct {
cache *RuntimeCacheService
images *ImageProxy
// Caches provider-chain results keyed by media kind/query/year. Per-instance
// so services with different provider config never share results.
lookupCache *scrapeLookupCache
// Serializes final sidecar replacement. Windows cannot rename over an
// existing file, and concurrent scrapes can target the same sidecar.
artworkWriteMu sync.Mutex
@@ -50,6 +54,7 @@ func NewScraperService(
return &ScraperService{
cfg: cfg, log: log, repo: repo,
tmdb: tmdb, bangumi: bangumi, thetvdb: thetvdb, fanart: fanart, adult: adultProvider, hub: hub,
lookupCache: newScrapeLookupCache(),
}
}