mirror of
https://github.com/truewhile/MeBox.git
synced 2026-10-06 21:36:37 +08:00
feat: add configurable scraper throttling
This commit is contained in:
@@ -251,6 +251,8 @@ func setDefaults(v *viper.Viper) {
|
|||||||
v.SetDefault("organizer.smart_classify", false)
|
v.SetDefault("organizer.smart_classify", false)
|
||||||
v.SetDefault("organizer.auto_after_download", false)
|
v.SetDefault("organizer.auto_after_download", false)
|
||||||
v.SetDefault("organize.scrape_after", false)
|
v.SetDefault("organize.scrape_after", false)
|
||||||
|
v.SetDefault("scrape.delay_min_ms", 250)
|
||||||
|
v.SetDefault("scrape.delay_max_ms", 500)
|
||||||
v.SetDefault("organizer.categories.chinese_movie", "华语电影")
|
v.SetDefault("organizer.categories.chinese_movie", "华语电影")
|
||||||
v.SetDefault("organizer.categories.animation_movie", "动画电影")
|
v.SetDefault("organizer.categories.animation_movie", "动画电影")
|
||||||
v.SetDefault("organizer.categories.foreign_movie", "外语电影")
|
v.SetDefault("organizer.categories.foreign_movie", "外语电影")
|
||||||
|
|||||||
@@ -94,6 +94,8 @@ func schemaHandler(_ *service.Container) gin.HandlerFunc {
|
|||||||
{"key": "scrape.auto_on_scan", "type": "toggle"},
|
{"key": "scrape.auto_on_scan", "type": "toggle"},
|
||||||
{"key": "scrape.providers", "type": "text"},
|
{"key": "scrape.providers", "type": "text"},
|
||||||
{"key": "scrape.language", "type": "text"},
|
{"key": "scrape.language", "type": "text"},
|
||||||
|
{"key": "scrape.delay_min_ms", "type": "number", "label": "刮削最小间隔毫秒"},
|
||||||
|
{"key": "scrape.delay_max_ms", "type": "number", "label": "刮削最大间隔毫秒"},
|
||||||
},
|
},
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
|
|||||||
@@ -13,6 +13,7 @@ package service
|
|||||||
|
|
||||||
import (
|
import (
|
||||||
"context"
|
"context"
|
||||||
|
"math/rand"
|
||||||
"path/filepath"
|
"path/filepath"
|
||||||
"regexp"
|
"regexp"
|
||||||
"strconv"
|
"strconv"
|
||||||
@@ -110,6 +111,12 @@ var multiWordNoise = []*regexp.Regexp{
|
|||||||
regexp.MustCompile(`(?i)\bohys[\s._-]*raws\b`),
|
regexp.MustCompile(`(?i)\bohys[\s._-]*raws\b`),
|
||||||
}
|
}
|
||||||
|
|
||||||
|
const (
|
||||||
|
defaultScrapeDelayMinMS = 250
|
||||||
|
defaultScrapeDelayMaxMS = 500
|
||||||
|
maxScrapeDelayMS = 5 * 60 * 1000
|
||||||
|
)
|
||||||
|
|
||||||
// CleanQuery converts a filename like "Inception.2010.1080p.BluRay.x264.mkv"
|
// CleanQuery converts a filename like "Inception.2010.1080p.BluRay.x264.mkv"
|
||||||
// into a TMDb-friendly title plus an optional year hint.
|
// into a TMDb-friendly title plus an optional year hint.
|
||||||
func CleanQuery(raw string) (title string, year int) {
|
func CleanQuery(raw string) (title string, year int) {
|
||||||
@@ -628,7 +635,15 @@ func (s *ScraperService) EnrichLibrary(ctx context.Context, libraryID string, re
|
|||||||
if s.mediaIsMatched(ctx, rows[i].ID) {
|
if s.mediaIsMatched(ctx, rows[i].ID) {
|
||||||
matched++
|
matched++
|
||||||
}
|
}
|
||||||
time.Sleep(250 * time.Millisecond) // ~4 RPS
|
if i < len(rows)-1 {
|
||||||
|
if delay := s.scrapeDelay(ctx); delay > 0 {
|
||||||
|
select {
|
||||||
|
case <-ctx.Done():
|
||||||
|
return matched, ctx.Err()
|
||||||
|
case <-time.After(delay):
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
}
|
}
|
||||||
s.hub.Publish("scrape", map[string]any{
|
s.hub.Publish("scrape", map[string]any{
|
||||||
"library_id": libraryID,
|
"library_id": libraryID,
|
||||||
@@ -639,6 +654,44 @@ func (s *ScraperService) EnrichLibrary(ctx context.Context, libraryID string, re
|
|||||||
return matched, nil
|
return matched, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func (s *ScraperService) scrapeDelay(ctx context.Context) time.Duration {
|
||||||
|
minMS := s.scrapeDelaySetting(ctx, "scrape.delay_min_ms", defaultScrapeDelayMinMS)
|
||||||
|
maxMS := s.scrapeDelaySetting(ctx, "scrape.delay_max_ms", defaultScrapeDelayMaxMS)
|
||||||
|
if minMS < 0 {
|
||||||
|
minMS = 0
|
||||||
|
}
|
||||||
|
if maxMS < 0 {
|
||||||
|
maxMS = 0
|
||||||
|
}
|
||||||
|
if minMS > maxScrapeDelayMS {
|
||||||
|
minMS = maxScrapeDelayMS
|
||||||
|
}
|
||||||
|
if maxMS > maxScrapeDelayMS {
|
||||||
|
maxMS = maxScrapeDelayMS
|
||||||
|
}
|
||||||
|
if maxMS < minMS {
|
||||||
|
maxMS = minMS
|
||||||
|
}
|
||||||
|
if maxMS == 0 {
|
||||||
|
return 0
|
||||||
|
}
|
||||||
|
if maxMS == minMS {
|
||||||
|
return time.Duration(minMS) * time.Millisecond
|
||||||
|
}
|
||||||
|
return time.Duration(minMS+rand.Intn(maxMS-minMS+1)) * time.Millisecond
|
||||||
|
}
|
||||||
|
|
||||||
|
func (s *ScraperService) scrapeDelaySetting(ctx context.Context, key string, fallback int) int {
|
||||||
|
if s == nil || s.repo == nil || s.repo.Setting == nil {
|
||||||
|
return fallback
|
||||||
|
}
|
||||||
|
value, err := s.repo.Setting.Get(ctx, key)
|
||||||
|
if err != nil || strings.TrimSpace(value) == "" {
|
||||||
|
return fallback
|
||||||
|
}
|
||||||
|
return parseIntSettingDefault(strings.TrimSpace(value), fallback)
|
||||||
|
}
|
||||||
|
|
||||||
func (s *ScraperService) mediaIsMatched(ctx context.Context, mediaID string) bool {
|
func (s *ScraperService) mediaIsMatched(ctx context.Context, mediaID string) bool {
|
||||||
var status string
|
var status string
|
||||||
err := s.repo.DB.WithContext(ctx).Model(&model.Media{}).
|
err := s.repo.DB.WithContext(ctx).Model(&model.Media{}).
|
||||||
|
|||||||
@@ -8,6 +8,7 @@ import (
|
|||||||
"path/filepath"
|
"path/filepath"
|
||||||
"strings"
|
"strings"
|
||||||
"testing"
|
"testing"
|
||||||
|
"time"
|
||||||
|
|
||||||
"github.com/glebarez/sqlite"
|
"github.com/glebarez/sqlite"
|
||||||
"go.uber.org/zap"
|
"go.uber.org/zap"
|
||||||
@@ -200,6 +201,38 @@ func TestManualEnrichLibraryRetriesNoMatchAndCountsRealMatches(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func TestScrapeDelayUsesSettings(t *testing.T) {
|
||||||
|
scraper, repos, closeServer := newTestScraper(t)
|
||||||
|
defer closeServer()
|
||||||
|
if err := repos.DB.AutoMigrate(&model.Setting{}); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
if got := scraper.scrapeDelay(t.Context()); got < 250*time.Millisecond || got > 500*time.Millisecond {
|
||||||
|
t.Fatalf("default scrapeDelay = %s, want 250-500ms", got)
|
||||||
|
}
|
||||||
|
|
||||||
|
if err := repos.Setting.Set(t.Context(), "scrape.delay_min_ms", "0"); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if err := repos.Setting.Set(t.Context(), "scrape.delay_max_ms", "0"); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if got := scraper.scrapeDelay(t.Context()); got != 0 {
|
||||||
|
t.Fatalf("disabled scrapeDelay = %s, want 0", got)
|
||||||
|
}
|
||||||
|
|
||||||
|
if err := repos.Setting.Set(t.Context(), "scrape.delay_min_ms", "800"); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if err := repos.Setting.Set(t.Context(), "scrape.delay_max_ms", "200"); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if got := scraper.scrapeDelay(t.Context()); got != 800*time.Millisecond {
|
||||||
|
t.Fatalf("normalized scrapeDelay = %s, want 800ms", got)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
func newTestScraper(t *testing.T) (*ScraperService, *repository.Container, func()) {
|
func newTestScraper(t *testing.T) (*ScraperService, *repository.Container, func()) {
|
||||||
t.Helper()
|
t.Helper()
|
||||||
|
|
||||||
|
|||||||
@@ -257,6 +257,22 @@ const GROUPS: SettingGroup[] = [
|
|||||||
type: 'text',
|
type: 'text',
|
||||||
placeholder: 'zh-CN',
|
placeholder: 'zh-CN',
|
||||||
},
|
},
|
||||||
|
{
|
||||||
|
key: 'scrape.delay_min_ms',
|
||||||
|
label: '刮削最小间隔毫秒',
|
||||||
|
type: 'number',
|
||||||
|
hint: '参考 nowen-video 的随机节流策略。批量刮削时两条媒体之间会随机等待,避免 TMDb / Bangumi / JavBus / JavDB 等源请求过快。',
|
||||||
|
defaultValue: '250',
|
||||||
|
placeholder: '250',
|
||||||
|
},
|
||||||
|
{
|
||||||
|
key: 'scrape.delay_max_ms',
|
||||||
|
label: '刮削最大间隔毫秒',
|
||||||
|
type: 'number',
|
||||||
|
hint: '如遇到站点限速、超时或 403,可提高到 2000-5000;填 0 可关闭批量刮削间隔。',
|
||||||
|
defaultValue: '500',
|
||||||
|
placeholder: '500',
|
||||||
|
},
|
||||||
{
|
{
|
||||||
key: 'scan.periodic_enabled',
|
key: 'scan.periodic_enabled',
|
||||||
label: '周期性整库重扫',
|
label: '周期性整库重扫',
|
||||||
|
|||||||
Reference in New Issue
Block a user