优化,排查项目问题

This commit is contained in:
truewhile
2026-09-05 12:34:17 +08:00
parent 7fa05391e1
commit 1407b9b5c4
85 changed files with 1940 additions and 638 deletions
+9
View File
@@ -6,6 +6,7 @@ import (
"errors"
"fmt"
"strings"
"time"
"github.com/glebarez/sqlite"
"go.uber.org/zap"
@@ -73,6 +74,14 @@ func configureConnectionPool(db *gorm.DB, cfg *config.Config) error {
if cfg.Database.MaxIdleConns > 0 {
sqlDB.SetMaxIdleConns(cfg.Database.MaxIdleConns)
}
// 连接生命周期:默认 0 意味着 Postgres 重启/故障切换后的陈旧连接
// 永不过期,首次复用才报错,运行期断连恢复慢且可能批量报错。
if isPostgres(db) {
sqlDB.SetConnMaxLifetime(time.Hour)
sqlDB.SetConnMaxIdleTime(10 * time.Minute)
} else if isSQLite(db) {
sqlDB.SetConnMaxLifetime(24 * time.Hour)
}
return nil
}
+1 -1
View File
@@ -9,7 +9,7 @@ const mediaSearchIndexSchemaVersion = 2
func ensureMediaSearchIndex(db *gorm.DB) error {
if err := ensureMediaSearchMetaTable(db); err != nil {
return nil
return err // meta 表创建失败必须上抛,不能静默掩盖
}
version := currentMediaSearchIndexVersion(db)
if version != mediaSearchIndexSchemaVersion {
+10 -4
View File
@@ -132,13 +132,19 @@ func ensureEmbyMountsCompatibility(db *gorm.DB) error {
return err
}
}
// 针对已有数据:如果存在多个 sort_order=0/NULL 的记录,按创建时间顺序赋予稳定递增的序号
// 针对已有数据:只给 sort_order=0/NULL 的行按创建时间补号(从现有
// 最大值之后递增),不能整表重排——此前无条件按 created_at 从 0 重新
// 编号,会把用户自定义的顺序覆盖掉。
var zeroCount int64
if err := db.Model(&model.EmbyMount{}).Where("sort_order = 0 OR sort_order IS NULL").Count(&zeroCount).Error; err == nil && zeroCount > 1 {
if err := db.Model(&model.EmbyMount{}).Where("sort_order = 0 OR sort_order IS NULL").Count(&zeroCount).Error; err == nil && zeroCount > 0 {
// max 只统计非 0 行:sort_order=0 与 NULL 同样视为“未分配”,
// 全部为 0 时从 0 开始编号(与迁移前的初始化语义一致)。
var maxOrder int
_ = db.Raw("SELECT COALESCE(MAX(sort_order), -1) FROM emby_mounts WHERE sort_order > 0").Scan(&maxOrder).Error
var mounts []model.EmbyMount
if err := db.Order("created_at asc, id asc").Find(&mounts).Error; err == nil {
if err := db.Where("sort_order = 0 OR sort_order IS NULL").Order("created_at asc, id asc").Find(&mounts).Error; err == nil {
for i, m := range mounts {
_ = db.Exec("UPDATE emby_mounts SET sort_order = ? WHERE id = ?", i, m.ID).Error
_ = db.Exec("UPDATE emby_mounts SET sort_order = ? WHERE id = ?", maxOrder+1+i, m.ID).Error
}
}
}
+31 -16
View File
@@ -50,28 +50,43 @@ func copyModelTables(src, target *gorm.DB, batchSize int) (map[string]int64, int
if modelType.Kind() != reflect.Ptr {
return tableCounts, totalCopied, fmt.Errorf("model %T is not a pointer", m)
}
sliceType := reflect.SliceOf(modelType.Elem())
slicePtr := reflect.New(sliceType)
if err := src.Unscoped().Find(slicePtr.Interface()).Error; err != nil {
return tableCounts, totalCopied, fmt.Errorf("read sqlite table %s: %w", table, err)
}
filtered := slicePtr.Elem()
var primaryKeySet map[string]struct{}
if targetCount > 0 {
primaryKeySet, err := targetPrimaryKeySet(target, table, primaryColumns)
primaryKeySet, err = targetPrimaryKeySet(target, table, primaryColumns)
if err != nil {
return tableCounts, totalCopied, err
}
filtered = filterRowsMissingInTarget(target, table, primaryColumns, filtered, primaryKeySet)
}
if filtered.Len() == 0 {
continue
// 分页流式读取:此前整表一次性 Find 进内存,media 表几十万行、
// 每行含 overview/genres 等长文本时可达数百 MB,迁移过程有 OOM
// 风险。源库在迁移期间是静态的,offset 分页安全。
const readBatch = 1000
copiedForTable := int64(0)
for offset := 0; ; offset += readBatch {
batchPtr := reflect.New(reflect.SliceOf(modelType.Elem()))
if err := src.Unscoped().Limit(readBatch).Offset(offset).Find(batchPtr.Interface()).Error; err != nil {
return tableCounts, totalCopied, fmt.Errorf("read sqlite table %s: %w", table, err)
}
batch := batchPtr.Elem()
if batch.Len() == 0 {
break
}
filtered := batch
if primaryKeySet != nil {
filtered = filterRowsMissingInTarget(target, table, primaryColumns, batch, primaryKeySet)
}
if filtered.Len() > 0 {
filteredPtr := reflect.New(filtered.Type())
filteredPtr.Elem().Set(filtered)
if err := target.Clauses(clause.OnConflict{DoNothing: true}).CreateInBatches(filteredPtr.Interface(), batchSize).Error; err != nil {
return tableCounts, totalCopied, fmt.Errorf("copy sqlite table %s: %w", table, err)
}
copiedForTable += int64(filtered.Len())
}
if batch.Len() < readBatch {
break
}
}
filteredPtr := reflect.New(filtered.Type())
filteredPtr.Elem().Set(filtered)
if err := target.Clauses(clause.OnConflict{DoNothing: true}).CreateInBatches(filteredPtr.Interface(), batchSize).Error; err != nil {
return tableCounts, totalCopied, fmt.Errorf("copy sqlite table %s: %w", table, err)
}
copiedForTable := int64(filtered.Len())
tableCounts[table] = copiedForTable
totalCopied += copiedForTable
}
+76 -24
View File
@@ -5,12 +5,21 @@ import (
"fmt"
"path/filepath"
"strings"
"sync"
"sync/atomic"
"time"
"gorm.io/gorm"
"github.com/truewhile/MeBox/internal/config"
)
// sqliteGateHoldLimit 是写闸持有者的最长合法持有时长。语句级写闸在 SQL
// 执行 panic 时 After 回调不会运行,令牌会泄漏并让后续所有写入永久等锁;
// 超过该时长的持有者按泄漏强制回收(60s 内单条写语句远未到,正常写路径
// 不受影响)。
const sqliteGateHoldLimit = 60 * time.Second
func installSQLiteWriteGate(db *gorm.DB) {
if db == nil {
return
@@ -22,15 +31,18 @@ func installSQLiteWriteGate(db *gorm.DB) {
if tx.Statement != nil && tx.Statement.Context != nil {
ctx = tx.Statement.Context
}
if err := gate.Lock(ctx); err != nil {
holder, err := gate.Lock(ctx)
if err != nil {
_ = tx.AddError(err)
return
}
tx.InstanceSet(lockedKey, struct{}{})
tx.InstanceSet(lockedKey, holder)
}
unlock := func(tx *gorm.DB) {
if _, ok := tx.InstanceGet(lockedKey); ok {
gate.Unlock()
if holder, ok := tx.InstanceGet(lockedKey); ok {
if h, ok := holder.(*sqliteGateHolder); ok {
gate.Unlock(h)
}
}
}
rawLock := func(tx *gorm.DB) {
@@ -64,38 +76,76 @@ func isReadOnlySQL(sql string) bool {
return false
}
// sqliteWriteGate serializes in-process SQLite writes while respecting the
// statement context, so request cancellation can break out of a queued write.
// sqliteWriteGate serializes in-process SQLite writes. 所有权令牌(而非裸
// 信号量)保证只有持有者本人能释放;持有超时按泄漏自动回收,避免一次
// panic 让进程的 SQLite 写入半永久性瘫痪。
type sqliteWriteGate struct {
ch chan struct{}
mu sync.Mutex
cond *sync.Cond
owner *sqliteGateHolder
}
type sqliteGateHolder struct {
id uint64
acquired time.Time
}
var sqliteGateHolderSeq atomic.Uint64
func newSQLiteWriteGate() *sqliteWriteGate {
return &sqliteWriteGate{ch: make(chan struct{}, 1)}
g := &sqliteWriteGate{}
g.cond = sync.NewCond(&g.mu)
return g
}
func (g *sqliteWriteGate) Lock(ctx context.Context) error {
select {
case g.ch <- struct{}{}:
return nil
default:
}
func (g *sqliteWriteGate) Lock(ctx context.Context) (*sqliteGateHolder, error) {
g.mu.Lock()
defer g.mu.Unlock()
if ctx == nil {
ctx = context.Background()
}
select {
case g.ch <- struct{}{}:
return nil
case <-ctx.Done():
return ctx.Err()
// ctx 取消时唤醒等待者(cond 无法感知 ctx,用旁路 goroutine 广播)。
if done := ctx.Done(); done != nil {
stop := make(chan struct{})
defer close(stop)
go func() {
select {
case <-done:
g.cond.Broadcast()
case <-stop:
}
}()
}
for {
if g.owner == nil {
holder := &sqliteGateHolder{
id: sqliteGateHolderSeq.Add(1),
acquired: time.Now(),
}
g.owner = holder
return holder, nil
}
if ctx.Err() != nil {
return nil, ctx.Err()
}
if time.Since(g.owner.acquired) > sqliteGateHoldLimit {
// 持有者疑似 panic 泄漏(After 回调未执行):强制回收。
g.owner = nil
g.cond.Broadcast()
continue
}
g.cond.Wait()
}
}
func (g *sqliteWriteGate) Unlock() {
select {
case <-g.ch:
default:
func (g *sqliteWriteGate) Unlock(h *sqliteGateHolder) {
g.mu.Lock()
defer g.mu.Unlock()
if h == nil || g.owner != h {
return
}
g.owner = nil
g.cond.Broadcast()
}
func buildSQLiteDSN(cfg *config.Config) string {
@@ -104,7 +154,9 @@ func buildSQLiteDSN(cfg *config.Config) string {
// keep as-is to respect user-provided relative paths.
dbPath = filepath.Clean(dbPath)
}
dsn := dbPath + "?_pragma=foreign_keys(1)"
// _txlock=immediate:事务以写锁开始。此前 deferred BEGIN 在并发事务
// 升级写锁时会绕过 busy_timeout 直接报 SQLITE_BUSY。
dsn := dbPath + "?_txlock=immediate&_pragma=foreign_keys(1)"
if cfg.Database.WALMode {
dsn += "&_pragma=journal_mode(WAL)&_pragma=synchronous(NORMAL)"
}