refactor: split modules and harden scraping workflows

This commit is contained in:
ShukeBta
2026-06-24 11:59:18 +08:00
parent efca3cbe69
commit 192f35d9fa
470 changed files with 47957 additions and 33839 deletions
+103 -6
View File
@@ -4,6 +4,7 @@ package handler
import (
"context"
"errors"
"io"
"net/http"
"github.com/gin-gonic/gin"
@@ -79,16 +80,99 @@ func imageProxyHandler(svc *service.Container) gin.HandlerFunc {
}
}
func cloudArtworkProxyHandler(svc *service.Container) gin.HandlerFunc {
return func(c *gin.Context) {
typ := c.Param("type")
ref := c.Query("ref")
if !service.IsAdminCloudConfigurable(typ) {
c.JSON(http.StatusBadRequest, gin.H{"error": "unsupported cloud provider"})
return
}
if ref == "" || !isCloudImageRef(ref) {
c.JSON(http.StatusBadRequest, gin.H{"error": "image ref required"})
return
}
if svc == nil || svc.ImageProxy == nil {
c.JSON(http.StatusServiceUnavailable, gin.H{"error": "image proxy unavailable"})
return
}
stableKey := typ + ":" + ref
if svc.ImageProxy.ServeCloudCached(c.Writer, c.Request, stableKey) {
return
}
if svc.StorageCfg == nil {
c.JSON(http.StatusServiceUnavailable, gin.H{"error": "cloud storage service unavailable"})
return
}
link, err := svc.StorageCfg.CloudResolve(c.Request.Context(), typ, ref, c.Request.UserAgent())
if err != nil {
c.JSON(http.StatusBadGateway, gin.H{"error": err.Error()})
return
}
if err := svc.ImageProxy.ServeCloudResolved(c.Request.Context(), c.Writer, c.Request, stableKey, link); err != nil {
c.JSON(http.StatusBadGateway, gin.H{"error": err.Error()})
return
}
}
}
type scrapeRequest struct {
EpisodeArtwork *bool `json:"episode_artwork"`
EpisodeImages *bool `json:"episode_images"`
RefreshMatched *bool `json:"refresh_matched"`
IncludeMatched *bool `json:"include_matched"`
}
func (r scrapeRequest) episodeArtworkOption() *bool {
if r.EpisodeImages != nil {
return r.EpisodeImages
}
return r.EpisodeArtwork
}
func (r scrapeRequest) includeMatchedOption() bool {
if r.IncludeMatched != nil {
return *r.IncludeMatched
}
if r.RefreshMatched != nil {
return *r.RefreshMatched
}
return false
}
func scrapeOptionsFromRequest(c *gin.Context, retryNoMatch bool) (service.ScrapeOptions, error) {
options := service.ScrapeOptions{RetryNoMatch: retryNoMatch}
if c.Request.Body == nil || c.Request.ContentLength == 0 {
return options, nil
}
var req scrapeRequest
if err := c.ShouldBindJSON(&req); err != nil {
if errors.Is(err, io.EOF) {
return options, nil
}
return options, err
}
options.EpisodeArtwork = req.episodeArtworkOption()
options.IncludeMatched = req.includeMatchedOption()
return options, nil
}
// scrapeOneHandler enriches a single media via the configured scraper chain.
func scrapeOneHandler(svc *service.Container) gin.HandlerFunc {
return func(c *gin.Context) {
options, err := scrapeOptionsFromRequest(c, true)
if err != nil {
c.JSON(http.StatusBadRequest, gin.H{"error": "invalid scrape options"})
return
}
options.IncludeMatched = true
m, err := svc.Repo.Media.FindByID(c.Request.Context(), c.Param("id"))
if err != nil || m == nil {
c.JSON(http.StatusNotFound, gin.H{"error": "not found"})
return
}
task := startScrapeHTTPTask(svc, "手动刮削媒体", m.Title, m.Path)
if err := svc.Scraper.EnrichOne(c.Request.Context(), m); err != nil {
if err := svc.Scraper.EnrichOneWithOptions(c.Request.Context(), m, options); err != nil {
finishHTTPTask(task, err, "scrape", "手动刮削媒体失败", nil, nil)
c.JSON(http.StatusInternalServerError, gin.H{"error": err.Error()})
return
@@ -103,10 +187,16 @@ func scrapeOneHandler(svc *service.Container) gin.HandlerFunc {
}
}
// scrapeLibraryHandler retries every pending/no_match media in a library.
// scrapeLibraryHandler manually refreshes every scrapeable row in a library.
func scrapeLibraryHandler(svc *service.Container) gin.HandlerFunc {
return func(c *gin.Context) {
libID := c.Param("id")
options, err := scrapeOptionsFromRequest(c, true)
if err != nil {
c.JSON(http.StatusBadRequest, gin.H{"error": "invalid scrape options"})
return
}
options.IncludeMatched = true
var task *service.TaskHandle
if lib, err := svc.Repo.Library.FindByID(c.Request.Context(), libID); err == nil && lib != nil {
task = startScrapeHTTPTask(svc, "手动刮削媒体库", lib.Name, lib.Path)
@@ -115,9 +205,16 @@ func scrapeLibraryHandler(svc *service.Container) gin.HandlerFunc {
}
// Run in the background so HTTP returns instantly; the WS hub
// pushes per-item progress on the "scrape" topic.
go func(libID string, task *service.TaskHandle) {
matched, err := svc.Scraper.EnrichLibrary(context.Background(), libID, true)
metrics := map[string]int64{"matched": int64(matched)}
go func(libID string, task *service.TaskHandle, options service.ScrapeOptions) {
result, err := svc.Scraper.EnrichLibraryDetailedWithOptions(context.Background(), libID, options)
metrics := map[string]int64{
"matched": int64(result.Matched),
"processed": int64(result.Processed),
"candidates": int64(result.Candidates),
}
if result.Failed > 0 {
metrics["errors"] = int64(result.Failed)
}
stage := "completed"
message := "手动刮削媒体库结束"
if err != nil {
@@ -125,7 +222,7 @@ func scrapeLibraryHandler(svc *service.Container) gin.HandlerFunc {
message = "手动刮削媒体库失败"
}
finishHTTPTask(task, err, stage, message, metrics, nil)
}(libID, task)
}(libID, task, options)
c.JSON(http.StatusAccepted, gin.H{"status": "scraping"})
}
}