mirror of
https://github.com/truewhile/MeBox.git
synced 2026-09-29 19:36:36 +08:00
0f30c34463
Backend
- service/ffprobe.go: thin ffprobe wrapper, parses duration / resolution /
codecs into a typed ProbeResult. 30s per-file timeout.
- service/tmdb.go: minimal TMDb provider (search/movie). Disabled when no
api key; supports tmdb_api_proxy / tmdb_image_proxy overrides for users
behind a firewall.
- service/scraper.go: filename cleaner (handles bracketed tags, scene
noise tokens, year extraction), per-row + per-library enrichment with a
4 RPS throttle and WS hub progress events. Unit-tested.
- service/scanner.go: now invokes ffprobe per file and kicks the TMDb
scraper in the background once a library scan finishes.
- service/transcoder.go: per-media ffmpeg HLS job manager; outputs
index.m3u8 + seg_NNNNN.ts under cache/hls/<id>; cancels jobs on
shutdown; publishes 'transcode' WS events.
- service/stream.go: serves HLS playlist (with 30s wait-for-ready) and
.ts segments with path-traversal protection. Adds Probe() helper used
by the admin 'reprobe' button.
- service/image_proxy.go: cached, host-allow-listed reverse proxy for
TMDb / Bangumi / Douban / Fanart / TheTVDB images so the SPA never
hits a CORS or GFW issue.
- service/playback.go: history upsert, favourites toggle, playlist CRUD
+ ordered items. RecentHistory joins with model.Media in one extra
query so the home page can render a 'Continue Watching' row.
- handler/streaming.go + handler/playback.go: REST endpoints for HLS,
image proxy, scrape (one + library), reprobe, history, favourites,
playlists.
- handler/handler.go: registers /api/hls/:id/{index.m3u8,:seg}, /api/img,
/api/history, /api/favourites/:id, /api/playlists/* with proper
auth/admin guards.
Frontend
- api/client.ts: imageURL() helper; hlsURL() endpoint; reuses the JWT in
a query parameter for <video src> and <img src>.
- api/playback.ts: typed helpers for history, favourites, playlists.
- components/MediaCard.tsx: optional 'progress' prop renders a thin
bottom progress bar, used by the new Continue Watching row.
- pages/HomePage.tsx: two rows (Continue Watching + Recently Added);
falls back to the empty-state hint when both are empty.
- pages/PlayerPage.tsx: hls.js (lazy-imported) with auto-fallback to
direct play; ?mode=hls|direct query toggle; resume position written
every 10s while playing.
- pages/MediaDetailPage.tsx: heart toggle + admin 'rescrape' / 'reprobe'
buttons + dedicated 'HLS 转码播放' CTA.
- pages/FavouritesPage.tsx, PlaylistsPage.tsx, PlaylistDetailPage.tsx:
new screens.
- components/Layout.tsx + App.tsx: sidebar links for Favourites and
Playlists; routes are now lazily code-split via React.lazy + Suspense
so the initial bundle stays at ~243 KB / 82 KB gzipped (hls.js is
fetched only on first HLS playback).
Verified: go build, go vet, go test (incl. CleanQuery cases) all pass;
frontend tsc -b && vite build emits 9 route chunks plus a deferred hls
chunk.
165 lines
5.0 KiB
Go
165 lines
5.0 KiB
Go
// Package service — scraper orchestrator.
|
|
//
|
|
// ScraperService takes a Media row and tries to enrich it with metadata
|
|
// from one or more providers (currently TMDb only). It is invoked at the
|
|
// end of every scan cycle for media items whose `scrape_status` is still
|
|
// "pending"; it can also be re-triggered manually from the admin UI.
|
|
//
|
|
// The orchestrator is deliberately stateless: it loops media → provider →
|
|
// repository, publishing scrape progress events to the WS hub.
|
|
package service
|
|
|
|
import (
|
|
"context"
|
|
"path/filepath"
|
|
"regexp"
|
|
"strconv"
|
|
"strings"
|
|
"time"
|
|
|
|
"go.uber.org/zap"
|
|
|
|
"github.com/ShukeBta/MediaStationGo/internal/config"
|
|
"github.com/ShukeBta/MediaStationGo/internal/model"
|
|
"github.com/ShukeBta/MediaStationGo/internal/repository"
|
|
)
|
|
|
|
// ScraperService coordinates metadata enrichment across providers.
|
|
type ScraperService struct {
|
|
cfg *config.Config
|
|
log *zap.Logger
|
|
repo *repository.Container
|
|
tmdb *TMDbProvider
|
|
hub *Hub
|
|
}
|
|
|
|
// NewScraperService is the constructor.
|
|
func NewScraperService(cfg *config.Config, log *zap.Logger, repo *repository.Container, tmdb *TMDbProvider, hub *Hub) *ScraperService {
|
|
return &ScraperService{cfg: cfg, log: log, repo: repo, tmdb: tmdb, hub: hub}
|
|
}
|
|
|
|
// yearPattern extracts a 4-digit year from a filename (1900-2099).
|
|
var yearPattern = regexp.MustCompile(`(?:^|[^\d])(19\d{2}|20\d{2})(?:[^\d]|$)`)
|
|
|
|
// noiseTokens are aggressively stripped from filenames before search.
|
|
// Keep in sync with nowen-video's filename_parser.go intent.
|
|
var noiseTokens = []string{
|
|
"1080p", "2160p", "4k", "720p", "480p",
|
|
"hdrip", "bluray", "blu-ray", "webrip", "web-dl", "web",
|
|
"x264", "x265", "h264", "h265", "hevc", "avc",
|
|
"hdr", "sdr", "dts", "ddp", "atmos", "aac", "ac3", "flac",
|
|
"remux", "extended", "uncut", "directors-cut", "directors_cut",
|
|
"hkfree", "yify", "rarbg", "ettv", "fgt",
|
|
}
|
|
|
|
// bracketedTag matches "[anything]" or "(anything)" segments, which are
|
|
// almost always release-group / encoder tags in scene filenames.
|
|
var bracketedTag = regexp.MustCompile(`[\[\(][^\]\)]*[\]\)]`)
|
|
|
|
// CleanQuery converts a filename like "Inception.2010.1080p.BluRay.x264.mkv"
|
|
// into a TMDb-friendly title plus an optional year hint.
|
|
func CleanQuery(raw string) (title string, year int) {
|
|
name := strings.TrimSuffix(filepath.Base(raw), filepath.Ext(raw))
|
|
lower := strings.ToLower(name)
|
|
|
|
// 1. Year first — bracketed years (1999) must survive the next step.
|
|
if m := yearPattern.FindStringSubmatch(lower); len(m) >= 2 {
|
|
if v, err := strconv.Atoi(m[1]); err == nil {
|
|
year = v
|
|
lower = strings.ReplaceAll(lower, m[1], " ")
|
|
}
|
|
}
|
|
|
|
// 2. Drop everything inside square / round brackets — those are tags.
|
|
lower = bracketedTag.ReplaceAllString(lower, " ")
|
|
|
|
for _, t := range noiseTokens {
|
|
lower = strings.ReplaceAll(lower, t, " ")
|
|
}
|
|
// collapse separators / spaces
|
|
for _, sep := range []string{".", "_", "-", "[", "]", "(", ")"} {
|
|
lower = strings.ReplaceAll(lower, sep, " ")
|
|
}
|
|
fields := strings.Fields(lower)
|
|
title = strings.Join(fields, " ")
|
|
return strings.TrimSpace(title), year
|
|
}
|
|
|
|
// EnrichOne runs the provider chain for a single media row.
|
|
func (s *ScraperService) EnrichOne(ctx context.Context, m *model.Media) error {
|
|
if s.tmdb == nil || !s.tmdb.Enabled() {
|
|
return nil
|
|
}
|
|
query := m.Title
|
|
if query == "" {
|
|
query, _ = CleanQuery(m.Path)
|
|
} else {
|
|
query, _ = CleanQuery(query)
|
|
}
|
|
year := m.Year
|
|
if year == 0 {
|
|
_, year = CleanQuery(filepath.Base(m.Path))
|
|
}
|
|
match, err := s.tmdb.SearchMovie(ctx, query, year)
|
|
if err != nil || match == nil {
|
|
return err
|
|
}
|
|
updates := map[string]any{
|
|
"title": match.Title,
|
|
"overview": match.Overview,
|
|
"poster_url": match.PosterURL,
|
|
"backdrop_url": match.BackdropURL,
|
|
"rating": match.Rating,
|
|
"year": match.Year,
|
|
"tmdb_id": match.TMDbID,
|
|
"scrape_status": "matched",
|
|
}
|
|
if err := s.repo.DB.Model(&model.Media{}).Where("id = ?", m.ID).
|
|
Updates(updates).Error; err != nil {
|
|
return err
|
|
}
|
|
s.hub.Publish("scrape", map[string]any{
|
|
"media_id": m.ID,
|
|
"title": match.Title,
|
|
"tmdb_id": match.TMDbID,
|
|
})
|
|
return nil
|
|
}
|
|
|
|
// EnrichLibrary runs the provider chain for every "pending" media in a
|
|
// library. It throttles to 4 RPS to stay below TMDb's rate limit and
|
|
// publishes a summary event when done.
|
|
func (s *ScraperService) EnrichLibrary(ctx context.Context, libraryID string) (int, error) {
|
|
if s.tmdb == nil || !s.tmdb.Enabled() {
|
|
return 0, nil
|
|
}
|
|
var rows []model.Media
|
|
q := s.repo.DB.Where("scrape_status = ?", "pending")
|
|
if libraryID != "" {
|
|
q = q.Where("library_id = ?", libraryID)
|
|
}
|
|
if err := q.Find(&rows).Error; err != nil {
|
|
return 0, err
|
|
}
|
|
matched := 0
|
|
for i := range rows {
|
|
select {
|
|
case <-ctx.Done():
|
|
return matched, ctx.Err()
|
|
default:
|
|
}
|
|
if err := s.EnrichOne(ctx, &rows[i]); err != nil {
|
|
s.log.Warn("enrich failed", zap.String("media", rows[i].ID), zap.Error(err))
|
|
continue
|
|
}
|
|
matched++
|
|
time.Sleep(250 * time.Millisecond) // ~4 RPS
|
|
}
|
|
s.hub.Publish("scrape", map[string]any{
|
|
"library_id": libraryID,
|
|
"finished": true,
|
|
"matched": matched,
|
|
})
|
|
return matched, nil
|
|
}
|