perf: 重构 FTS 索引为 rowid 寻址+触发器维护,根治重启 CPU 占满与扫描卡死

- FTS5 普通列(含 UNINDEXED)不支持索引查找,旧版按 media_id 做
  NOT EXISTS/DELETE 全是整表扫描:启动回填 O(N^2) 烧 CPU 数小时,
  扫描时每个 upsert 一次全表扫描,均隔着全局写锁拖死登录(125s 超时)
- v2 布局:FTS rowid 与 media.rowid 对齐,由 INSERT/UPDATE/DELETE
  触发器实时维护;回填只剩 rowid 点查兜底,且去掉 ORDER BY
- 顺带修复刮削直写 Updates() 后新标题搜不到的问题(触发器覆盖)
- 写门改为 context 感知:长写语句不再让登录请求无限期排队
- warmMediaSearchIndex 延迟 30s 错峰启动
- library_scan 周期任务跳过云盘库(由 cloud_sync 夜间窗口负责),
  首轮延迟从 15s 改为等满一个周期,重启不再立即全量扫描风暴
- prune 改为只取 id/path 并按 500 条批量删除,缩短写锁占用
This commit is contained in:
ShukeBta
2026-06-12 06:41:14 +00:00
parent 8d1d1f18a8
commit 1568ae127a
6 changed files with 187 additions and 77 deletions
+38 -16
View File
@@ -1491,13 +1491,20 @@ func (s *ScannerService) mediaPathExists(ctx context.Context, path string) bool
}
func (s *ScannerService) pruneMissingMedia(ctx context.Context, libraryID string, seen map[string]struct{}) (int64, error) {
var rows []model.Media
// 只取 id/path,并把删除按批提交:此前整表载入完整 Media 结构体、
// 每行一条 DELETE,大库 prune 既费内存又长期占用写锁。
var rows []struct {
ID string
Path string
}
if err := s.repo.DB.WithContext(ctx).
Model(&model.Media{}).
Select("id, path").
Where("library_id = ?", libraryID).
Find(&rows).Error; err != nil {
return 0, err
}
var removed int64
stale := make([]string, 0)
for _, row := range rows {
if row.Path == "" {
continue
@@ -1510,9 +1517,26 @@ func (s *ScannerService) pruneMissingMedia(ctx context.Context, libraryID string
} else if !os.IsNotExist(err) {
continue
}
res := s.repo.DB.WithContext(ctx).
Where("id = ?", row.ID).
Delete(&model.Media{})
stale = append(stale, row.ID)
}
return s.deleteMediaByIDs(ctx, stale, false)
}
// deleteMediaByIDs removes media rows in fixed-size batches so each write
// transaction stays short and the global write gate is released frequently.
func (s *ScannerService) deleteMediaByIDs(ctx context.Context, ids []string, hard bool) (int64, error) {
const batch = 500
var removed int64
for i := 0; i < len(ids); i += batch {
end := i + batch
if end > len(ids) {
end = len(ids)
}
q := s.repo.DB.WithContext(ctx)
if hard {
q = q.Unscoped()
}
res := q.Where("id IN ?", ids[i:end]).Delete(&model.Media{})
if res.Error != nil {
return removed, res.Error
}
@@ -1522,27 +1546,25 @@ func (s *ScannerService) pruneMissingMedia(ctx context.Context, libraryID string
}
func (s *ScannerService) pruneMissingCloudMedia(ctx context.Context, libraryID string, seen map[string]struct{}) (int64, error) {
var rows []model.Media
var rows []struct {
ID string
Path string
}
if err := s.repo.DB.WithContext(ctx).
Model(&model.Media{}).
Select("id, path").
Where("library_id = ? AND path LIKE ?", libraryID, "cloud://%").
Find(&rows).Error; err != nil {
return 0, err
}
var removed int64
stale := make([]string, 0)
for _, row := range rows {
if _, ok := seen[row.Path]; ok {
continue
}
res := s.repo.DB.WithContext(ctx).
Unscoped().
Where("id = ?", row.ID).
Delete(&model.Media{})
if res.Error != nil {
return removed, res.Error
}
removed += res.RowsAffected
stale = append(stale, row.ID)
}
return removed, nil
return s.deleteMediaByIDs(ctx, stale, true)
}
func parseCloudLibraryPath(raw string) (typ, dirID string, ok bool) {
+15 -1
View File
@@ -134,7 +134,14 @@ func (s *SchedulerService) Start(ctx context.Context) {
},
}
for _, j := range s.jobs {
go s.loop(ctx, j)
initialDelay := 15 * time.Second
if j.name == "library_scan" {
// 重启后不立即整库重扫:更新/重启窗口恰是登录高峰,启动
// 15 秒即全量扫描曾把 CPU/磁盘打满导致无法登录。首轮等满
// 一个完整周期再跑,平时的每小时节奏不变。
initialDelay = j.interval
}
go s.loopWithInitialDelay(ctx, j, initialDelay)
}
}
@@ -259,6 +266,13 @@ func (s *SchedulerService) jobScanLibraries(ctx context.Context) error {
if !l.Enabled {
continue
}
if _, ok := ParseCloudLibraryMount(l.Path); ok {
// 云盘库由 cloud_sync 任务在夜间窗口低频同步;周期性整库
// 重扫只面向本地磁盘库。否则十几个云盘库每小时全量遍历
// 会把 CPU/网络长期吃满,还会占住唯一的云扫描槽位,让
// 手动扫描看起来一直"卡死"在排队。
continue
}
if _, err := s.scanner.ScanLibrary(ctx, l.ID); err != nil {
s.log.Warn("scheduled scan failed",
zap.String("library", l.ID), zap.Error(err))
+7
View File
@@ -268,6 +268,13 @@ func (c *Container) warmMediaSearchIndex(ctx context.Context) {
if c == nil || c.Repo == nil || c.Repo.Media == nil {
return
}
// 错峰:FTS 正常由 media 表触发器实时维护,回填只是升级或异常后的
// 兜底。先让登录、首页等关键路径跑起来,再开始后台补索引。
select {
case <-ctx.Done():
return
case <-time.After(30 * time.Second):
}
const batchSize = 1000
const pause = 100 * time.Millisecond
total := int64(0)