mirror of
https://github.com/truewhile/MeBox.git
synced 2026-09-29 11:36:36 +08:00
feat: add configurable scraper throttling
This commit is contained in:
@@ -251,6 +251,8 @@ func setDefaults(v *viper.Viper) {
|
||||
v.SetDefault("organizer.smart_classify", false)
|
||||
v.SetDefault("organizer.auto_after_download", false)
|
||||
v.SetDefault("organize.scrape_after", false)
|
||||
v.SetDefault("scrape.delay_min_ms", 250)
|
||||
v.SetDefault("scrape.delay_max_ms", 500)
|
||||
v.SetDefault("organizer.categories.chinese_movie", "华语电影")
|
||||
v.SetDefault("organizer.categories.animation_movie", "动画电影")
|
||||
v.SetDefault("organizer.categories.foreign_movie", "外语电影")
|
||||
|
||||
@@ -94,6 +94,8 @@ func schemaHandler(_ *service.Container) gin.HandlerFunc {
|
||||
{"key": "scrape.auto_on_scan", "type": "toggle"},
|
||||
{"key": "scrape.providers", "type": "text"},
|
||||
{"key": "scrape.language", "type": "text"},
|
||||
{"key": "scrape.delay_min_ms", "type": "number", "label": "刮削最小间隔毫秒"},
|
||||
{"key": "scrape.delay_max_ms", "type": "number", "label": "刮削最大间隔毫秒"},
|
||||
},
|
||||
},
|
||||
{
|
||||
|
||||
@@ -13,6 +13,7 @@ package service
|
||||
|
||||
import (
|
||||
"context"
|
||||
"math/rand"
|
||||
"path/filepath"
|
||||
"regexp"
|
||||
"strconv"
|
||||
@@ -110,6 +111,12 @@ var multiWordNoise = []*regexp.Regexp{
|
||||
regexp.MustCompile(`(?i)\bohys[\s._-]*raws\b`),
|
||||
}
|
||||
|
||||
const (
|
||||
defaultScrapeDelayMinMS = 250
|
||||
defaultScrapeDelayMaxMS = 500
|
||||
maxScrapeDelayMS = 5 * 60 * 1000
|
||||
)
|
||||
|
||||
// CleanQuery converts a filename like "Inception.2010.1080p.BluRay.x264.mkv"
|
||||
// into a TMDb-friendly title plus an optional year hint.
|
||||
func CleanQuery(raw string) (title string, year int) {
|
||||
@@ -628,7 +635,15 @@ func (s *ScraperService) EnrichLibrary(ctx context.Context, libraryID string, re
|
||||
if s.mediaIsMatched(ctx, rows[i].ID) {
|
||||
matched++
|
||||
}
|
||||
time.Sleep(250 * time.Millisecond) // ~4 RPS
|
||||
if i < len(rows)-1 {
|
||||
if delay := s.scrapeDelay(ctx); delay > 0 {
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return matched, ctx.Err()
|
||||
case <-time.After(delay):
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
s.hub.Publish("scrape", map[string]any{
|
||||
"library_id": libraryID,
|
||||
@@ -639,6 +654,44 @@ func (s *ScraperService) EnrichLibrary(ctx context.Context, libraryID string, re
|
||||
return matched, nil
|
||||
}
|
||||
|
||||
func (s *ScraperService) scrapeDelay(ctx context.Context) time.Duration {
|
||||
minMS := s.scrapeDelaySetting(ctx, "scrape.delay_min_ms", defaultScrapeDelayMinMS)
|
||||
maxMS := s.scrapeDelaySetting(ctx, "scrape.delay_max_ms", defaultScrapeDelayMaxMS)
|
||||
if minMS < 0 {
|
||||
minMS = 0
|
||||
}
|
||||
if maxMS < 0 {
|
||||
maxMS = 0
|
||||
}
|
||||
if minMS > maxScrapeDelayMS {
|
||||
minMS = maxScrapeDelayMS
|
||||
}
|
||||
if maxMS > maxScrapeDelayMS {
|
||||
maxMS = maxScrapeDelayMS
|
||||
}
|
||||
if maxMS < minMS {
|
||||
maxMS = minMS
|
||||
}
|
||||
if maxMS == 0 {
|
||||
return 0
|
||||
}
|
||||
if maxMS == minMS {
|
||||
return time.Duration(minMS) * time.Millisecond
|
||||
}
|
||||
return time.Duration(minMS+rand.Intn(maxMS-minMS+1)) * time.Millisecond
|
||||
}
|
||||
|
||||
func (s *ScraperService) scrapeDelaySetting(ctx context.Context, key string, fallback int) int {
|
||||
if s == nil || s.repo == nil || s.repo.Setting == nil {
|
||||
return fallback
|
||||
}
|
||||
value, err := s.repo.Setting.Get(ctx, key)
|
||||
if err != nil || strings.TrimSpace(value) == "" {
|
||||
return fallback
|
||||
}
|
||||
return parseIntSettingDefault(strings.TrimSpace(value), fallback)
|
||||
}
|
||||
|
||||
func (s *ScraperService) mediaIsMatched(ctx context.Context, mediaID string) bool {
|
||||
var status string
|
||||
err := s.repo.DB.WithContext(ctx).Model(&model.Media{}).
|
||||
|
||||
@@ -8,6 +8,7 @@ import (
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/glebarez/sqlite"
|
||||
"go.uber.org/zap"
|
||||
@@ -200,6 +201,38 @@ func TestManualEnrichLibraryRetriesNoMatchAndCountsRealMatches(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestScrapeDelayUsesSettings(t *testing.T) {
|
||||
scraper, repos, closeServer := newTestScraper(t)
|
||||
defer closeServer()
|
||||
if err := repos.DB.AutoMigrate(&model.Setting{}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
if got := scraper.scrapeDelay(t.Context()); got < 250*time.Millisecond || got > 500*time.Millisecond {
|
||||
t.Fatalf("default scrapeDelay = %s, want 250-500ms", got)
|
||||
}
|
||||
|
||||
if err := repos.Setting.Set(t.Context(), "scrape.delay_min_ms", "0"); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := repos.Setting.Set(t.Context(), "scrape.delay_max_ms", "0"); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if got := scraper.scrapeDelay(t.Context()); got != 0 {
|
||||
t.Fatalf("disabled scrapeDelay = %s, want 0", got)
|
||||
}
|
||||
|
||||
if err := repos.Setting.Set(t.Context(), "scrape.delay_min_ms", "800"); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := repos.Setting.Set(t.Context(), "scrape.delay_max_ms", "200"); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if got := scraper.scrapeDelay(t.Context()); got != 800*time.Millisecond {
|
||||
t.Fatalf("normalized scrapeDelay = %s, want 800ms", got)
|
||||
}
|
||||
}
|
||||
|
||||
func newTestScraper(t *testing.T) (*ScraperService, *repository.Container, func()) {
|
||||
t.Helper()
|
||||
|
||||
|
||||
@@ -257,6 +257,22 @@ const GROUPS: SettingGroup[] = [
|
||||
type: 'text',
|
||||
placeholder: 'zh-CN',
|
||||
},
|
||||
{
|
||||
key: 'scrape.delay_min_ms',
|
||||
label: '刮削最小间隔毫秒',
|
||||
type: 'number',
|
||||
hint: '参考 nowen-video 的随机节流策略。批量刮削时两条媒体之间会随机等待,避免 TMDb / Bangumi / JavBus / JavDB 等源请求过快。',
|
||||
defaultValue: '250',
|
||||
placeholder: '250',
|
||||
},
|
||||
{
|
||||
key: 'scrape.delay_max_ms',
|
||||
label: '刮削最大间隔毫秒',
|
||||
type: 'number',
|
||||
hint: '如遇到站点限速、超时或 403,可提高到 2000-5000;填 0 可关闭批量刮削间隔。',
|
||||
defaultValue: '500',
|
||||
placeholder: '500',
|
||||
},
|
||||
{
|
||||
key: 'scan.periodic_enabled',
|
||||
label: '周期性整库重扫',
|
||||
|
||||
Reference in New Issue
Block a user