mirror of
https://github.com/truewhile/MeBox.git
synced 2026-09-29 19:36:36 +08:00
4b747c74ca
Audit-driven port from the original Python MediaStation. Eight major
subsystems that were absent from the Go rewrite are now in place,
each with its own service, handler, frontend page and smoke-test
assertions.
Backend services
- service/crypto.go: AES-256-GCM encrypt/decrypt for at-rest secrets
keyed off the JWT secret. Legacy plaintext rows pass through
unchanged for smooth upgrades. Unit-tested.
- service/api_config.go: third-party provider config (TMDb, Bangumi,
TheTVDB, Fanart, Douban, OpenAI). Seeds defaults on first run.
Encrypts api_key on write, returns masked 'abc1****wxyz' projection.
- service/duplicate.go: sparse-sample MD5 (head + middle + tail, 1MiB
each, plus file-size suffix) duplicate finder. Picks 'best' primary
(matched > size > id) and marks others is_duplicate=true.
- service/filemanager.go: server-side allow-listed file browser used
by the library-path picker. Strict path-traversal protection.
- service/dlna.go: real SSDP M-SEARCH discovery + AVTransport
SetAVTransportURI/Play SOAP cast. 30 s discovery cache.
- service/scheduler.go: 3 recurring background jobs (library_scan
60min, transcode_cleanup 24h, recycle_purge 24h with 30-day
cutoff). Status + run-now endpoints.
- service/cache_cleanup.go: walkAndPrune helper used by scheduler.
- service/storage.go: DB-only disk-usage breakdown by library and by
container format.
- service/emby_compat.go: read-only Emby/Jellyfin shim
(System/Info, Users, Users/x/Views, Items, PlaybackInfo) so Infuse
/ VidHub / Kodi can browse MediaStationGo libraries.
Model updates
- Media: new strm_url (302 redirect target), file_hash, is_duplicate,
duplicate_of fields.
- APIConfig: new table for encrypted provider secrets.
- AutoMigrate registers APIConfig.
Stream layer
- StreamService.ServeFile now redirects 302 to strm_url when set so
WebDAV / Alist / S3 / HTTP direct links work transparently.
Handlers + routes
- Authed: GET /files, GET /storage, GET /dlna/devices, POST /dlna/cast,
PUT/DELETE /media/:id/strm, POST /strm/import,
POST /duplicates/{scan,unmark}.
- Admin: GET/PUT/DELETE /admin/api-configs/:provider,
GET /admin/scheduler, POST /admin/scheduler/:name/run.
- New /emby/* group: System/Info, Users, Users/:userId/Views,
Users/:userId/Items, Items/:id/PlaybackInfo (auth-required).
Frontend pages (lazy-loaded, 7 new chunks)
- DlnaPage: device list + media picker + cast button.
- FileManagerPage: root selector + breadcrumb + sortable listing.
- APIConfigsPage: per-provider card with masked-key editor.
- StoragePage: usage tiles + per-library bars + per-container grid.
- DuplicatesPage: scan form + grouped report with primary highlight.
- SchedulerPage: live job table with run-now button (5s refresh).
- Sidebar reorganised: 自动化 group adds DLNA, 管理 group adds
存储 / 文件浏览 / 重复文件 / 定时任务 / API 配置.
Smoke test additions (all admin-only)
- api-configs seeded with 6 providers
- api-config encrypted in db (sqlite3 enc:v1: prefix check)
- storage breakdown
- file browser lists library root + rejects /etc (path traversal)
- dlna devices endpoint
- scheduler exposes 3 jobs + run library_scan
- emby /System/Info + /Users/{x}/Views
- strm set + stream 302 + strm clear
- duplicate scan
Verified: go build, go vet, go test (incl. new TestCrypto* suite + the
existing TestParseEpisode/TestCleanQuery/TestSrtToVTT/TestStripASSTags/
TestBuildFFmpegArgs); tsc -b && vite build emits 28 route chunks plus
the deferred hls chunk; main bundle 253 KB / 85 KB gzipped; smoke test
PASS=42 / FAIL=0.
236 lines
6.3 KiB
Go
236 lines
6.3 KiB
Go
// Package service — duplicate-file finder.
|
|
//
|
|
// DuplicateService computes a sparse-sample MD5 (head + middle + tail,
|
|
// 1 MiB each, plus the file size to break collisions) for every media
|
|
// file and groups identical hashes into "duplicate sets". The first row
|
|
// (preferring scraped + larger files) is kept as the primary; the rest
|
|
// get is_duplicate = true and duplicate_of pointing at the primary.
|
|
//
|
|
// Why sparse: a full hash on a 50 GB Blu-ray remux takes minutes; the
|
|
// 3-window 3 MiB sample is enough to differentiate real-world copies
|
|
// while finishing per-file in well under a second.
|
|
package service
|
|
|
|
import (
|
|
"context"
|
|
"crypto/md5"
|
|
"encoding/hex"
|
|
"errors"
|
|
"fmt"
|
|
"io"
|
|
"os"
|
|
"sort"
|
|
|
|
"go.uber.org/zap"
|
|
|
|
"github.com/ShukeBta/MediaStationGo/internal/model"
|
|
"github.com/ShukeBta/MediaStationGo/internal/repository"
|
|
)
|
|
|
|
const sampleSize = 1 << 20 // 1 MiB per sample window
|
|
|
|
// DuplicateService is the entry point for the duplicate finder.
|
|
type DuplicateService struct {
|
|
log *zap.Logger
|
|
repo *repository.Container
|
|
hub *Hub
|
|
}
|
|
|
|
// NewDuplicateService is the constructor.
|
|
func NewDuplicateService(log *zap.Logger, repo *repository.Container, hub *Hub) *DuplicateService {
|
|
return &DuplicateService{log: log, repo: repo, hub: hub}
|
|
}
|
|
|
|
// Group describes one set of duplicates returned by Detect.
|
|
type Group struct {
|
|
Hash string `json:"hash"`
|
|
Primary model.Media `json:"primary"`
|
|
Duplicates []model.Media `json:"duplicates"`
|
|
}
|
|
|
|
// Report is the summary the React UI displays.
|
|
type Report struct {
|
|
TotalScanned int `json:"total_scanned"`
|
|
GroupsFound int `json:"groups_found"`
|
|
ItemsMarked int `json:"items_marked"`
|
|
Groups []Group `json:"groups"`
|
|
}
|
|
|
|
// Detect walks every media row in the given library (or all libraries
|
|
// when libraryID is empty), computes a hash for the ones missing it,
|
|
// then groups by hash and marks duplicates in the DB.
|
|
func (d *DuplicateService) Detect(ctx context.Context, libraryID string) (*Report, error) {
|
|
var rows []model.Media
|
|
q := d.repo.DB.WithContext(ctx).Model(&model.Media{})
|
|
if libraryID != "" {
|
|
q = q.Where("library_id = ?", libraryID)
|
|
}
|
|
if err := q.Find(&rows).Error; err != nil {
|
|
return nil, err
|
|
}
|
|
|
|
rep := &Report{TotalScanned: len(rows)}
|
|
totalToHash := 0
|
|
for i := range rows {
|
|
if rows[i].FileHash == "" && rows[i].Path != "" {
|
|
totalToHash++
|
|
}
|
|
}
|
|
|
|
hashed := 0
|
|
for i := range rows {
|
|
select {
|
|
case <-ctx.Done():
|
|
return rep, ctx.Err()
|
|
default:
|
|
}
|
|
if rows[i].FileHash != "" || rows[i].Path == "" {
|
|
continue
|
|
}
|
|
h, err := SparseFileHash(rows[i].Path)
|
|
if err != nil {
|
|
d.log.Debug("hash failed", zap.String("path", rows[i].Path), zap.Error(err))
|
|
continue
|
|
}
|
|
rows[i].FileHash = h
|
|
if err := d.repo.DB.WithContext(ctx).
|
|
Model(&model.Media{}).
|
|
Where("id = ?", rows[i].ID).
|
|
Update("file_hash", h).Error; err != nil {
|
|
d.log.Warn("hash persist failed", zap.Error(err))
|
|
}
|
|
hashed++
|
|
if d.hub != nil && totalToHash > 0 {
|
|
d.hub.Publish("duplicate", map[string]any{
|
|
"hashed": hashed,
|
|
"total": totalToHash,
|
|
"current": rows[i].Title,
|
|
})
|
|
}
|
|
}
|
|
|
|
// Group rows by file_hash.
|
|
groups := make(map[string][]model.Media)
|
|
for _, r := range rows {
|
|
if r.FileHash == "" {
|
|
continue
|
|
}
|
|
groups[r.FileHash] = append(groups[r.FileHash], r)
|
|
}
|
|
|
|
for hash, group := range groups {
|
|
if len(group) < 2 {
|
|
continue
|
|
}
|
|
primary := pickPrimary(group)
|
|
dupes := make([]model.Media, 0, len(group)-1)
|
|
for _, m := range group {
|
|
if m.ID == primary.ID {
|
|
continue
|
|
}
|
|
dupes = append(dupes, m)
|
|
if err := d.repo.DB.WithContext(ctx).
|
|
Model(&model.Media{}).
|
|
Where("id = ?", m.ID).
|
|
Updates(map[string]any{
|
|
"is_duplicate": true,
|
|
"duplicate_of": primary.ID,
|
|
}).Error; err != nil {
|
|
d.log.Warn("dup mark failed", zap.Error(err))
|
|
continue
|
|
}
|
|
rep.ItemsMarked++
|
|
}
|
|
rep.Groups = append(rep.Groups, Group{
|
|
Hash: hash,
|
|
Primary: primary,
|
|
Duplicates: dupes,
|
|
})
|
|
}
|
|
rep.GroupsFound = len(rep.Groups)
|
|
if d.hub != nil {
|
|
d.hub.Publish("duplicate", map[string]any{
|
|
"finished": true,
|
|
"groups": rep.GroupsFound,
|
|
"marked": rep.ItemsMarked,
|
|
})
|
|
}
|
|
return rep, nil
|
|
}
|
|
|
|
// Unmark clears the is_duplicate flag for every row in the given library
|
|
// (or all when libraryID is empty). Useful when the operator deletes the
|
|
// physical duplicates manually.
|
|
func (d *DuplicateService) Unmark(ctx context.Context, libraryID string) (int64, error) {
|
|
q := d.repo.DB.WithContext(ctx).Model(&model.Media{}).Where("is_duplicate = ?", true)
|
|
if libraryID != "" {
|
|
q = q.Where("library_id = ?", libraryID)
|
|
}
|
|
res := q.Updates(map[string]any{"is_duplicate": false, "duplicate_of": ""})
|
|
return res.RowsAffected, res.Error
|
|
}
|
|
|
|
// pickPrimary picks the "best" media row to keep: prefer scraped > size > id.
|
|
func pickPrimary(group []model.Media) model.Media {
|
|
sort.SliceStable(group, func(i, j int) bool {
|
|
ai, aj := group[i].ScrapeStatus == "matched", group[j].ScrapeStatus == "matched"
|
|
if ai != aj {
|
|
return ai
|
|
}
|
|
if group[i].SizeBytes != group[j].SizeBytes {
|
|
return group[i].SizeBytes > group[j].SizeBytes
|
|
}
|
|
return group[i].ID < group[j].ID
|
|
})
|
|
return group[0]
|
|
}
|
|
|
|
// SparseFileHash computes the head+mid+tail MD5 of a file, suffixed with
|
|
// the file size so two files that happen to collide on the sample window
|
|
// but differ in length are still distinguishable.
|
|
func SparseFileHash(path string) (string, error) {
|
|
if path == "" {
|
|
return "", errors.New("empty path")
|
|
}
|
|
f, err := os.Open(path)
|
|
if err != nil {
|
|
return "", err
|
|
}
|
|
defer f.Close()
|
|
st, err := f.Stat()
|
|
if err != nil {
|
|
return "", err
|
|
}
|
|
size := st.Size()
|
|
h := md5.New()
|
|
if size <= int64(sampleSize)*3 {
|
|
if _, err := io.Copy(h, f); err != nil {
|
|
return "", err
|
|
}
|
|
return fmt.Sprintf("%s-%d", hex.EncodeToString(h.Sum(nil)), size), nil
|
|
}
|
|
buf := make([]byte, sampleSize)
|
|
// head
|
|
if _, err := io.ReadFull(f, buf); err != nil {
|
|
return "", err
|
|
}
|
|
h.Write(buf)
|
|
// middle
|
|
if _, err := f.Seek(size/2-int64(sampleSize)/2, io.SeekStart); err != nil {
|
|
return "", err
|
|
}
|
|
if _, err := io.ReadFull(f, buf); err != nil {
|
|
return "", err
|
|
}
|
|
h.Write(buf)
|
|
// tail
|
|
if _, err := f.Seek(size-int64(sampleSize), io.SeekStart); err != nil {
|
|
return "", err
|
|
}
|
|
if _, err := io.ReadFull(f, buf); err != nil {
|
|
return "", err
|
|
}
|
|
h.Write(buf)
|
|
return fmt.Sprintf("%s-%d", hex.EncodeToString(h.Sum(nil)), size), nil
|
|
}
|