fix(scrape): 单集名不再覆盖整剧 original_name 修复合集被拆集 + 媒体库页新增单库修复重刮

- 根因: 单集刮削把 TMDB episode.Name 写入 original_name(应为整剧原名),
  导致同剧每集 original_name 不同, 合集分组键回退到标题时被拆成多集无法合并。
  移除该覆盖, 保留单集 overview/剧照/评分/时长回填(不影响合集键)。
- 新增 RepairAndRescrapeLibrary 服务方法 + POST /admin/libraries/:id/repair-rescrape
  路由 + 媒体库页「修复+重刮本库」按钮, 可按库单独触发修复(回填占位符外部 ID)并重刮,
  既修空 ID 又借重刮覆盖此前被单集名污染的 original_name。
This commit is contained in:
ShukeBta
2026-06-19 15:00:00 +08:00
parent f873997d27
commit 9b4884359b
7 changed files with 105 additions and 6 deletions
+28
View File
@@ -36,3 +36,31 @@ func repairAndRescrapeAllHandler(svc *service.Container) gin.HandlerFunc {
c.JSON(http.StatusAccepted, gin.H{"status": "started"})
}
}
// repairAndRescrapeLibraryHandler 触发"单库修复+重刮":只对路径参数指定的
// 媒体库回填占位符外部 ID 并重刮, 不影响其它库。
//
// 路由: POST /api/admin/libraries/:id/repair-rescrape (需 admin)
// 异步执行, 立即返回 202;通过 WS hub "scrape" topic 推送进度。
func repairAndRescrapeLibraryHandler(svc *service.Container) gin.HandlerFunc {
return func(c *gin.Context) {
libraryID := c.Param("id")
task := startScrapeHTTPTask(svc, "媒体库修复并重刮", "", "")
go func() {
result, err := svc.RepairAndRescrapeLibrary(context.Background(), libraryID)
metrics := map[string]int64{
"repaired": int64(result.Repaired),
"libraries": int64(result.Libraries),
"matched": int64(result.Matched),
}
stage := "completed"
message := "媒体库修复并重刮完成"
if err != nil {
stage = "scrape"
message = "媒体库修复并重刮失败"
}
finishHTTPTask(task, err, stage, message, metrics, nil)
}()
c.JSON(http.StatusAccepted, gin.H{"status": "started"})
}
}
+2
View File
@@ -89,6 +89,8 @@ func registerAdminRoutes(api *gin.RouterGroup, cfg *config.Config, svc *service.
// 全库修复+重刮:从路径占位符回填缺失外部 ID,然后批量重刮整库。
admin.POST("/media/repair-rescrape", repairAndRescrapeAllHandler(svc))
// 单库修复+重刮:只对指定媒体库回填占位符外部 ID 并重刮。
admin.POST("/libraries/:id/repair-rescrape", repairAndRescrapeLibraryHandler(svc))
// API key management (encrypted at rest).
admin.GET("/api-configs", listAPIConfigsHandler(svc))
+40 -1
View File
@@ -13,7 +13,9 @@ import (
// RepairCloudPathMetadata backfills external IDs from media paths such as
// "Movie (2025) {tmdb-123}" so existing placeholder rows can be scraped
// without requiring another successful filesystem or cloud provider traversal.
func (c *Container) RepairCloudPathMetadata(ctx context.Context) (int, error) {
//
// 传入 libraryID 时只修复该媒体库的行;为空则修复全库。
func (c *Container) RepairCloudPathMetadata(ctx context.Context, libraryID ...string) (int, error) {
if c == nil || c.Repo == nil || c.Repo.DB == nil {
return 0, nil
}
@@ -33,6 +35,9 @@ func (c *Container) RepairCloudPathMetadata(ctx context.Context) (int, error) {
"LOWER(path) LIKE ?",
}, " OR ")+")",
"%tmdb%", "%tmdbid%", "%douban%", "%db%", "%bangumi%", "%bgm%", "%thetvdb%", "%tvdb%")
if len(libraryID) > 0 && strings.TrimSpace(libraryID[0]) != "" {
query = query.Where("library_id = ?", strings.TrimSpace(libraryID[0]))
}
err := query.FindInBatches(&rows, 500, func(_ *gorm.DB, _ int) error {
for _, row := range rows {
@@ -166,3 +171,37 @@ func (c *Container) RepairAndRescrapeAllLibraries(ctx context.Context) (RepairAn
}
return result, nil
}
// RepairAndRescrapeLibrary 修复并重刮单个媒体库:先从该库媒体路径中的占位符
// 回填缺失/错误的外部 ID(重置相关行 scrape_status=pending),再对该库重刮
// (含 no_match 重试)。用于「按媒体库」单独触发修复,不影响其它库。
func (c *Container) RepairAndRescrapeLibrary(ctx context.Context, libraryID string) (RepairAndRescrapeResult, error) {
var result RepairAndRescrapeResult
libraryID = strings.TrimSpace(libraryID)
if c == nil || c.Repo == nil || c.Repo.DB == nil || libraryID == "" {
return result, nil
}
repaired, err := c.RepairCloudPathMetadata(ctx, libraryID)
if err != nil {
return result, err
}
result.Repaired = repaired
if c.Scraper == nil {
return result, nil
}
result.Libraries = 1
// retryNoMatch=true:连之前匹配失败的也再试一次,因为这次可能已回填到正确 ID。
matched, err := c.Scraper.EnrichLibrary(ctx, libraryID, true)
if err != nil {
return result, err
}
result.Matched = matched
if c.Log != nil {
c.Log.Info("repair and rescrape library done",
zap.String("library", libraryID),
zap.Int("repaired", result.Repaired),
zap.Int("matched", result.Matched))
}
return result, nil
}
+6 -3
View File
@@ -546,9 +546,12 @@ func (s *ScraperService) applyProviderMatch(ctx context.Context, m *model.Media,
zap.Error(err))
} else if episode != nil {
episodeUpdates := map[string]any{}
if strings.TrimSpace(episode.Name) != "" {
episodeUpdates["original_name"] = strings.TrimSpace(episode.Name)
}
// 注意: 不要把 episode.Name(单集名,如"觉醒"/"Pilot"/"第1集")写入
// original_name —— 该字段是「整剧原名」,是合集分组键的回退依据。
// 若每集都写成各自的单集名,同一部剧的各集 original_name 互不相同,
// 会被前端 getSeriesKey / 后端 mediaVersionGroupKey 拆成多个独立卡片,
// 导致同剧无法合并成合集。单集名属于单集信息,这里只回填不影响合集
// 分组的单集字段(简介/剧照/评分/时长)。
if strings.TrimSpace(episode.Overview) != "" {
episodeUpdates["overview"] = strings.TrimSpace(episode.Overview)
}
+9 -2
View File
@@ -489,8 +489,9 @@ func TestEnrichOneWritesTMDbEpisodeMetadata(t *testing.T) {
if err := repos.DB.First(&got, "id = ?", media.ID).Error; err != nil {
t.Fatal(err)
}
if got.OriginalName != "任务代号: 猫" || got.Overview != "单集剧情" {
t.Fatalf("episode metadata not saved: original=%q overview=%q", got.OriginalName, got.Overview)
// 单集专属信息(简介/剧照/评分/时长)应回填到该集行。
if got.Overview != "单集剧情" {
t.Fatalf("episode overview not saved: overview=%q", got.Overview)
}
if !strings.HasSuffix(got.BackdropURL, "/images/w500/still.jpg") || got.DurationSec != 24*60 {
t.Fatalf("episode still/runtime not saved: backdrop=%q duration=%d", got.BackdropURL, got.DurationSec)
@@ -498,6 +499,11 @@ func TestEnrichOneWritesTMDbEpisodeMetadata(t *testing.T) {
if got.Rating < 9.09 || got.Rating > 9.11 {
t.Fatalf("episode rating = %v, want 9.1", got.Rating)
}
// original_name 必须保持「整剧原名」,绝不能被单集名(任务代号: 猫)覆盖,
// 否则同剧每集 original_name 不同会导致合集被拆成多集无法合并。
if got.OriginalName != "SPY×FAMILY" {
t.Fatalf("original_name should stay series-level, got %q (episode name must not overwrite it)", got.OriginalName)
}
}
func TestEnrichOneRejectsWrongYearMatchFromSeriesFolder(t *testing.T) {
@@ -701,6 +707,7 @@ func newTestScraper(t *testing.T) (*ScraperService, *repository.Container, func(
"results": []map[string]any{{
"id": 12345,
"name": "间谍过家家",
"original_name": "SPY×FAMILY",
"overview": "测试简介",
"poster_path": "/poster.jpg",
"backdrop_path": "/backdrop.jpg",