Fix manual scrape apply for cloud media

This commit is contained in:
ShukeBta
2026-06-18 00:27:26 +08:00
parent 0291ab8194
commit e4b1488e11
4 changed files with 117 additions and 14 deletions
+14 -2
View File
@@ -1,8 +1,10 @@
package handler
import (
"context"
"net/http"
"strings"
"time"
"github.com/gin-gonic/gin"
@@ -14,6 +16,8 @@ type manualScrapeApplyReq struct {
Match service.ManualScrapeRequest `json:"match"`
}
const manualScrapeApplyTimeout = 5 * time.Minute
func manualScrapeSearchHandler(svc *service.Container) gin.HandlerFunc {
return func(c *gin.Context) {
m, err := svc.Repo.Media.FindByID(c.Request.Context(), c.Param("id"))
@@ -43,7 +47,9 @@ func manualScrapeApplyOneHandler(svc *service.Container) gin.HandlerFunc {
c.JSON(http.StatusBadRequest, gin.H{"error": err.Error()})
return
}
media, err := svc.Scraper.ApplyManualMatch(c.Request.Context(), c.Param("id"), req)
applyCtx, cancel := manualScrapeApplyContext(c)
defer cancel()
media, err := svc.Scraper.ApplyManualMatch(applyCtx, c.Param("id"), req)
if err != nil {
c.JSON(http.StatusInternalServerError, gin.H{"error": err.Error()})
return
@@ -64,10 +70,12 @@ func manualScrapeApplyBatchHandler(svc *service.Container) gin.HandlerFunc {
c.JSON(http.StatusBadRequest, gin.H{"error": "media_ids required"})
return
}
applyCtx, cancel := manualScrapeApplyContext(c)
defer cancel()
applied := 0
errorsOut := make([]string, 0)
for _, id := range ids {
if _, err := svc.Scraper.ApplyManualMatch(c.Request.Context(), id, req.Match); err != nil {
if _, err := svc.Scraper.ApplyManualMatch(applyCtx, id, req.Match); err != nil {
errorsOut = append(errorsOut, id+": "+err.Error())
continue
}
@@ -81,6 +89,10 @@ func manualScrapeApplyBatchHandler(svc *service.Container) gin.HandlerFunc {
}
}
func manualScrapeApplyContext(c *gin.Context) (context.Context, context.CancelFunc) {
return context.WithTimeout(context.WithoutCancel(c.Request.Context()), manualScrapeApplyTimeout)
}
func compactManualScrapeIDs(values []string) []string {
seen := map[string]struct{}{}
out := make([]string, 0, len(values))
+26 -10
View File
@@ -70,6 +70,8 @@ func (s *ScraperService) SetNotifyChannels(notify *NotifyChannelService) {
// yearPattern extracts a 4-digit year (1900-2099).
var yearPattern = regexp.MustCompile(`(?:^|[^\d])(19\d{2}|20\d{2})(?:[^\d]|$)`)
var tmdbDetailsTimeout = 8 * time.Second
// noiseTokens are stripped before search.
var noiseTokens = []string{
// 视频规格
@@ -429,24 +431,43 @@ func (s *ScraperService) applyProviderMatch(ctx context.Context, m *model.Media,
updates["languages"] = strings.Join(match.Languages, ",")
}
// Fetch extended metadata (languages, countries, genres) from TMDb
if err := s.repo.DB.Model(&model.Media{}).Where("id = ?", m.ID).
Updates(updates).Error; err != nil {
return err
}
// Fetch extended metadata after the selected match is already saved.
// Manual cloud/batch applies must not fail just because an optional provider
// details request is slow or unavailable.
if match.TMDbID > 0 && s.tmdb != nil && s.tmdb.Enabled() {
mediaType := s.determineMediaType(lib, match)
details, err := s.tmdb.GetDetails(ctx, match.TMDbID, mediaType)
detailCtx, cancel := context.WithTimeout(context.WithoutCancel(ctx), tmdbDetailsTimeout)
details, err := s.tmdb.GetDetails(detailCtx, match.TMDbID, mediaType)
cancel()
if err != nil {
s.log.Warn("failed to get details from tmdb",
zap.Int("tmdb_id", match.TMDbID),
zap.String("type", mediaType),
zap.Error(err))
} else if details != nil {
detailUpdates := map[string]any{}
if len(details.Languages) > 0 {
updates["languages"] = strings.Join(details.Languages, ",")
detailUpdates["languages"] = strings.Join(details.Languages, ",")
}
if len(details.Countries) > 0 {
updates["countries"] = strings.Join(details.Countries, ",")
detailUpdates["countries"] = strings.Join(details.Countries, ",")
}
if len(details.Genres) > 0 {
updates["genres"] = strings.Join(details.Genres, ",")
detailUpdates["genres"] = strings.Join(details.Genres, ",")
}
if len(detailUpdates) > 0 {
if err := s.repo.DB.Model(&model.Media{}).Where("id = ?", m.ID).
Updates(detailUpdates).Error; err != nil {
s.log.Warn("failed to save tmdb extended metadata",
zap.String("media_id", m.ID),
zap.Int("tmdb_id", match.TMDbID),
zap.Error(err))
}
}
s.log.Debug("enrich: saved extended metadata",
zap.String("media_id", m.ID),
@@ -455,11 +476,6 @@ func (s *ScraperService) applyProviderMatch(ctx context.Context, m *model.Media,
zap.Strings("genres", details.Genres))
}
}
if err := s.repo.DB.Model(&model.Media{}).Where("id = ?", m.ID).
Updates(updates).Error; err != nil {
return err
}
cloudMedia := isCloudMediaPath(m.Path) || (lib != nil && isCloudMediaPath(lib.Path))
if !cloudMedia {
if refreshed, err := s.repo.Media.FindByID(ctx, m.ID); err == nil && refreshed != nil {
+73
View File
@@ -61,6 +61,79 @@ func TestManualRequestMatchFallsBackToCandidatePayload(t *testing.T) {
}
}
func TestApplyManualMatchSavesSelectedCloudMatchWhenDetailsSlow(t *testing.T) {
oldTimeout := tmdbDetailsTimeout
tmdbDetailsTimeout = 20 * time.Millisecond
defer func() { tmdbDetailsTimeout = oldTimeout }()
upstream := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
if r.URL.Path != "/movie/77" {
http.NotFound(w, r)
return
}
select {
case <-r.Context().Done():
return
case <-time.After(time.Second):
_ = json.NewEncoder(w).Encode(map[string]any{
"id": 77,
"title": "Slow Details",
})
}
}))
defer upstream.Close()
db, err := gorm.Open(sqlite.Open("file::memory:?cache=shared"), &gorm.Config{})
if err != nil {
t.Fatal(err)
}
if err := db.AutoMigrate(&model.Library{}, &model.Series{}, &model.Media{}); err != nil {
t.Fatal(err)
}
repos := repository.New(db)
cfg := &config.Config{}
cfg.Secrets.TMDbAPIKey = "test-key"
cfg.Secrets.TMDbAPIProxy = upstream.URL
log := zap.NewNop()
scraper := NewScraperService(cfg, log, repos, NewTMDbProvider(cfg, log, nil), nil, nil, nil, NewHub(log))
lib := model.Library{Name: "OpenList · Movies", Path: "cloud://openlist/Movies", Type: "movie", Enabled: true}
if err := repos.DB.Create(&lib).Error; err != nil {
t.Fatal(err)
}
media := model.Media{
LibraryID: lib.ID,
Title: "bad cloud title",
Path: "cloud://openlist/Movies/Bad.Title.2026.mkv",
ScrapeStatus: "pending",
}
if err := repos.DB.Create(&media).Error; err != nil {
t.Fatal(err)
}
start := time.Now()
if _, err := scraper.ApplyManualMatch(t.Context(), media.ID, ManualScrapeRequest{
Source: "manual",
MediaType: "movie",
Title: "Correct Cloud Movie",
TMDbID: 77,
Year: 2026,
}); err != nil {
t.Fatal(err)
}
if elapsed := time.Since(start); elapsed > 500*time.Millisecond {
t.Fatalf("manual apply waited for optional details: %s", elapsed)
}
var got model.Media
if err := repos.DB.First(&got, "id = ?", media.ID).Error; err != nil {
t.Fatal(err)
}
if got.Title != "Correct Cloud Movie" || got.ScrapeStatus != "matched" || got.TMDbID != 77 {
t.Fatalf("manual cloud match was not saved: title=%q status=%q tmdb=%d", got.Title, got.ScrapeStatus, got.TMDbID)
}
}
func TestScrapeQueryCandidatesPreferSeriesFolderAndCJKTitle(t *testing.T) {
lib := &model.Library{
Path: `F:\downloads\国产剧`,
+4 -2
View File
@@ -81,8 +81,10 @@ export const mediaAPI = {
.then((r) => r.data.items),
applyManualScrape: (id: string, match: ManualScrapeCandidate) =>
api.post<Media>(`/media/${id}/scrape/apply`, match).then((r) => r.data),
api.post<Media>(`/media/${id}/scrape/apply`, match, { timeout: LONG_REQUEST_TIMEOUT }).then((r) => r.data),
applyManualScrapeBatch: (mediaIDs: string[], match: ManualScrapeCandidate) =>
api.post<{ applied: number; errors?: string[] }>('/media/scrape/apply', { media_ids: mediaIDs, match }).then((r) => r.data),
api
.post<{ applied: number; errors?: string[] }>('/media/scrape/apply', { media_ids: mediaIDs, match }, { timeout: BATCH_REQUEST_TIMEOUT })
.then((r) => r.data),
}