From 9cc21a47c5eba3b9d2ab861610814395a5333d9a Mon Sep 17 00:00:00 2001 From: truewhile <779943132@qq.com> Date: Wed, 30 Sep 2026 16:55:52 +0800 Subject: [PATCH] =?UTF-8?q?feat(reader):=20P2=20=E6=94=B6=E5=B0=BE?= =?UTF-8?q?=E2=80=94=E2=80=94=E5=86=92=E7=83=9F=20CLI=E3=80=81=E4=B9=A6?= =?UTF-8?q?=E6=9E=B6=E6=AD=A3=E6=96=87=E9=93=BE=E8=B7=AF=E3=80=81=E6=9B=BF?= =?UTF-8?q?=E6=8D=A2=E5=87=80=E5=8C=96=E8=A7=84=E5=88=99?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - SmokeChain 结构化全链路冒烟(搜索→详情→目录→正文,分阶段失败定位), SmokeSource 支持不落库直测;Debug 接口改为结构化日志输出 - cmd/reader-smoke:批量书源冒烟 CLI(-file/-url 导入、并发、 各阶段通过率汇总、-json 完整报告) - GetContentForBook:书架维度正文(章节缓存 + 书源 replaceRegex + 用户替换净化规则),前端阅读器切换到该接口 - 替换净化规则 CRUD(后端 + /reader/replace 前端页),用户规则 按 scope/excludeScope 书名过滤、正则带匹配超时防回溯挂死 - rule.ApplyUserReplace 导出(regexp2 MatchTimeout) --- cmd/reader-smoke/main.go | 197 +++++++++++++ docs/reader-plan.md | 2 +- internal/handler/reader_routes.go | 67 +++++ internal/service/reader/reader.go | 325 ++++++++++++++++++--- internal/service/reader/reader_test.go | 6 +- internal/service/reader/rule/regexan.go | 27 ++ web/src/api/reader.ts | 26 +- web/src/pages/reader/ReaderReplacePage.tsx | 215 ++++++++++++++ web/src/pages/reader/ReaderRoutes.tsx | 2 + web/src/pages/reader/ReaderSourcesPage.tsx | 5 +- web/src/pages/reader/ReaderViewPage.tsx | 21 +- 11 files changed, 828 insertions(+), 65 deletions(-) create mode 100644 cmd/reader-smoke/main.go create mode 100644 web/src/pages/reader/ReaderReplacePage.tsx diff --git a/cmd/reader-smoke/main.go b/cmd/reader-smoke/main.go new file mode 100644 index 0000000..c6e6822 --- /dev/null +++ b/cmd/reader-smoke/main.go @@ -0,0 +1,197 @@ +// reader-smoke 是阅读书源兼容性冒烟工具: +// 对批量书源逐个跑「搜索 → 详情 → 目录 → 正文」全链路,输出兼容率报告。 +// +// 用法: +// +// go run ./cmd/reader-smoke -file sources.json -key 斗破苍穹 -c 8 +// go run ./cmd/reader-smoke -url https://example.com/sources.json -json > report.json +package main + +import ( + "context" + "encoding/json" + "flag" + "fmt" + "net/http" + "os" + "strings" + "sync" + "time" + + "go.uber.org/zap" + + "github.com/truewhile/MeBox/internal/config" + "github.com/truewhile/MeBox/internal/helper" + "github.com/truewhile/MeBox/internal/repository" + "github.com/truewhile/MeBox/internal/service/reader" +) + +func main() { + file := flag.String("file", "", "书源文件路径(JSON 数组/对象/Base64/每行一个)") + urlFlag := flag.String("url", "", "书源网络地址(与 -file 二选一)") + key := flag.String("key", "斗破苍穹", "搜索关键词") + concurrency := flag.Int("c", 4, "并发数") + timeout := flag.Int("timeout", 90, "单源全链路超时(秒)") + jsonOut := flag.Bool("json", false, "输出完整 JSON 报告(追加在汇总后)") + flag.Parse() + + payload := "" + switch { + case *file != "": + data, err := os.ReadFile(*file) + if err != nil { + fatal("读取文件失败: %v", err) + } + payload = string(data) + case *urlFlag != "": + client := helper.NewSiteHTTPClient(30, true) + req, err := http.NewRequest("GET", *urlFlag, nil) + if err != nil { + fatal("构造请求失败: %v", err) + } + for k, v := range helper.HTTPHeaderPresets() { + req.Header.Set(k, v) + } + resp, err := client.Do(req) + if err != nil { + fatal("拉取书源失败: %v", err) + } + defer resp.Body.Close() + var sb strings.Builder + buf := make([]byte, 32*1024) + for { + n, err := resp.Body.Read(buf) + sb.Write(buf[:n]) + if err != nil { + break + } + } + payload = sb.String() + default: + fatal("需要 -file 或 -url 指定书源来源") + } + + sources := reader.ParseSourcePayload(payload) + if len(sources) == 0 { + fatal("未从输入中识别到书源") + } + + svc := reader.NewReaderService(&config.Config{}, zap.NewNop(), &repository.Container{}) + ctx := context.Background() + + results := make([]*reader.SmokeChainResult, len(sources)) + sem := make(chan struct{}, max(1, *concurrency)) + var wg sync.WaitGroup + for i, raw := range sources { + wg.Add(1) + sem <- struct{}{} + go func(i int, raw string) { + defer wg.Done() + defer func() { <-sem }() + ctxSrc, cancel := context.WithTimeout(ctx, time.Duration(*timeout)*time.Second) + defer cancel() + res := svc.SmokeSource(ctxSrc, raw, *key) + results[i] = res + status := "✓" + if !res.OK { + status = "✗" + } + fmt.Fprintf(os.Stderr, "%s %-24s [%s] hits=%d chapters=%d content=%d %s\n", + status, truncate(res.SourceName, 24), stageCN(res), res.SearchHits, res.Chapters, res.ContentLen, res.Error) + }(i, raw) + } + wg.Wait() + + // 汇总 + var searchOK, infoOK, tocOK, contentOK, allOK int + failedAt := map[string]int{} + for _, r := range results { + if r == nil { + continue + } + switch r.FailedAt { + case "": + allOK++ + searchOK++ + infoOK++ + tocOK++ + contentOK++ + case "search": + failedAt["search"]++ + case "info": + searchOK++ + failedAt["info"]++ + case "toc": + searchOK++ + infoOK++ + failedAt["toc"]++ + case "content": + searchOK++ + infoOK++ + tocOK++ + failedAt["content"]++ + } + } + n := len(results) + pct := func(v int) string { + if n == 0 { + return "0%" + } + return fmt.Sprintf("%.1f%%", float64(v)/float64(n)*100) + } + fmt.Printf("\n==== 冒烟报告 ====\n") + fmt.Printf("书源总数: %d 关键词: %s\n", n, *key) + fmt.Printf("搜索通过: %d (%s)\n", searchOK, pct(searchOK)) + fmt.Printf("详情通过: %d (%s)\n", infoOK, pct(infoOK)) + fmt.Printf("目录通过: %d (%s)\n", tocOK, pct(tocOK)) + fmt.Printf("正文通过: %d (%s)\n", contentOK, pct(contentOK)) + fmt.Printf("全链路通过: %d (%s)\n", allOK, pct(allOK)) + for _, stage := range []string{"search", "info", "toc", "content"} { + if failedAt[stage] > 0 { + fmt.Printf(" 失败于 %s: %d\n", stageCN(&reader.SmokeChainResult{FailedAt: stage}), failedAt[stage]) + } + } + + if *jsonOut { + out, err := json.MarshalIndent(results, "", " ") + if err != nil { + fatal("序列化报告失败: %v", err) + } + fmt.Println(string(out)) + } +} + +func stageCN(r *reader.SmokeChainResult) string { + switch r.FailedAt { + case "": + return "完成" + case "search": + return "搜索" + case "info": + return "详情" + case "toc": + return "目录" + case "content": + return "正文" + case "parse": + return "解析" + default: + return r.FailedAt + } +} + +func truncate(s string, n int) string { + rs := []rune(strings.TrimSpace(s)) + if len(rs) <= n { + if len(rs) == 0 { + return "(未命名)" + } + return string(rs) + } + return string(rs[:n]) + "…" +} + +func fatal(format string, args ...any) { + fmt.Fprintf(os.Stderr, "reader-smoke: "+format+"\n", args...) + os.Exit(1) +} diff --git a/docs/reader-plan.md b/docs/reader-plan.md index 69c40e2..277b80f 100644 --- a/docs/reader-plan.md +++ b/docs/reader-plan.md @@ -119,7 +119,7 @@ web/src/ |---|---|---| | P0 引擎地基 ✅ | 规则引擎核心(四分析器 + 规则拆分/组合/变量)+ AnalyzeUrl v1(GET/POST/charset/headers/变量/页码模式)+ 表结构 + 书源导入/管理 API + 搜索/详情/目录/正文/书架/进度/调试 API | 已完成:`internal/service/reader/rule/`(规则引擎,~2800 行,对齐 AnalyzeRule/AnalyzeByJSoup/AnalyzeByJSonPath/AnalyzeByXPath/AnalyzeByRegex/AnalyzeUrl/RuleAnalyzer)+ 服务层 + `/api/reader/*` 路由 + 单测/端到端测试全绿 | | P1 文本源全链路 + 首页切换 | 搜索聚合(WS 进度)/详情/目录/正文(nextContentUrl 合并、缓存)+ 前端首页切换、书架、搜索、详情、文本阅读器 v1(阅读器样式仿 legado:9 宫格点击、主题、翻页动画) | 用纯规则型文本源完成「搜书→加入→阅读」全流程 | -| P2 JS 与兼容率爬坡 | goja 接入 + `java.*` 桥分批实现 + 加解密族 + URL 完整选项 + replaceRegex + 替换规则管理 + 书源调试页(仿 legado 逐条日志流式输出)+ 冒烟指标报告 | 主流公开文本源通过率显著提升,形成回归基线 | +| P2 JS 与兼容率爬坡 ✅ | goja 接入 + `java.*` 桥(网络/编解码/摘要/对称加密全家桶/规则回调,函数名对齐 JsExtensions)+ URL 规则 JS(analyzeJs/{{}}/js/bodyJs)+ cookie jar + 用户替换净化规则(含正则超时保护)+ 替换净化页 + 结构化冒烟链路(SmokeChain)+ `cmd/reader-smoke` 冒烟 CLI | JS 源可用;冒烟 CLI 跑公开书源集出各阶段通过率报告 | | P3 音频源 | 播放列表解析、音频代理(带 UA/Referer)、音频播放器页(仿 ReadAloudDialog 布局:上一章/播放/下一章/定时/倍速)、进度记忆 | 音频源可听 | | P4 漫画/图片源 | 图片列表解析(含翻页)、图片代理接入磁盘缓存、漫画阅读器双模式(MangaMenu:顶栏+底部胶囊)、预加载 | 漫画源可看 | | P5 体验完善 | 换源(ChangeBookSourceDialog 四档排序)、追更(定时刷新目录 + 缓存清理)、发现页(exploreUrl 标签条)、阅读器高级设置(页眉页脚提示、点击区域自定义)、书源编辑器六 Tab、备份导出;可选:本地 TXT/EPUB | 完整体验 | diff --git a/internal/handler/reader_routes.go b/internal/handler/reader_routes.go index ecd2e6e..ec76255 100644 --- a/internal/handler/reader_routes.go +++ b/internal/handler/reader_routes.go @@ -3,6 +3,7 @@ package handler import ( "net/http" + "strconv" "github.com/gin-gonic/gin" @@ -36,9 +37,13 @@ func registerReaderRoutes(authed *gin.RouterGroup, svc *service.Container) { g.PUT("/books/:id/progress", readerSaveProgressHandler(svc)) g.GET("/books/:id/chapters", readerListChaptersHandler(svc)) g.POST("/books/:id/chapters", readerReplaceChaptersHandler(svc)) + g.GET("/books/:id/content", readerBookContentHandler(svc)) // 替换净化规则 g.GET("/replace-rules", readerListReplaceRulesHandler(svc)) + g.POST("/replace-rules", readerCreateReplaceRuleHandler(svc)) + g.PATCH("/replace-rules/:id", readerUpdateReplaceRuleHandler(svc)) + g.DELETE("/replace-rules/:id", readerDeleteReplaceRuleHandler(svc)) } func readerListSourcesHandler(svc *service.Container) gin.HandlerFunc { @@ -279,3 +284,65 @@ func readerListReplaceRulesHandler(svc *service.Container) gin.HandlerFunc { c.JSON(http.StatusOK, gin.H{"rules": rules}) } } + +func readerCreateReplaceRuleHandler(svc *service.Container) gin.HandlerFunc { + var body reader.ReplaceRuleInput + return func(c *gin.Context) { + if err := c.ShouldBindJSON(&body); err != nil { + c.JSON(http.StatusBadRequest, gin.H{"error": err.Error()}) + return + } + userID := c.GetString(middleware.CtxUserID) + rule, err := svc.Reader.CreateReplaceRule(c.Request.Context(), userID, body) + if err != nil { + c.JSON(http.StatusBadRequest, gin.H{"error": err.Error()}) + return + } + c.JSON(http.StatusOK, rule) + } +} + +func readerUpdateReplaceRuleHandler(svc *service.Container) gin.HandlerFunc { + var body reader.ReplaceRuleInput + return func(c *gin.Context) { + if err := c.ShouldBindJSON(&body); err != nil { + c.JSON(http.StatusBadRequest, gin.H{"error": err.Error()}) + return + } + userID := c.GetString(middleware.CtxUserID) + if err := svc.Reader.UpdateReplaceRule(c.Request.Context(), userID, c.Param("id"), body); err != nil { + c.JSON(http.StatusBadRequest, gin.H{"error": err.Error()}) + return + } + c.JSON(http.StatusOK, gin.H{"ok": true}) + } +} + +func readerDeleteReplaceRuleHandler(svc *service.Container) gin.HandlerFunc { + return func(c *gin.Context) { + userID := c.GetString(middleware.CtxUserID) + if err := svc.Reader.DeleteReplaceRule(c.Request.Context(), userID, c.Param("id")); err != nil { + c.JSON(http.StatusInternalServerError, gin.H{"error": err.Error()}) + return + } + c.JSON(http.StatusOK, gin.H{"ok": true}) + } +} + +// readerBookContentHandler 书架维度正文(服务端应用替换净化规则)。 +func readerBookContentHandler(svc *service.Container) gin.HandlerFunc { + return func(c *gin.Context) { + chapter, err := strconv.Atoi(c.DefaultQuery("chapter", "0")) + if err != nil { + c.JSON(http.StatusBadRequest, gin.H{"error": "chapter 参数需为整数"}) + return + } + userID := c.GetString(middleware.CtxUserID) + content, err := svc.Reader.GetContentForBook(c.Request.Context(), userID, c.Param("id"), chapter) + if err != nil { + c.JSON(http.StatusBadRequest, gin.H{"error": err.Error()}) + return + } + c.JSON(http.StatusOK, content) + } +} diff --git a/internal/service/reader/reader.go b/internal/service/reader/reader.go index 2ec846c..4a72a99 100644 --- a/internal/service/reader/reader.go +++ b/internal/service/reader/reader.go @@ -65,7 +65,7 @@ func (s *ReaderService) ImportSources(ctx context.Context, text string) (int, er } text = body } - sources := parseSourcePayload(text) + sources := ParseSourcePayload(text) if len(sources) == 0 { return 0, fmt.Errorf("未识别到有效书源(支持 JSON 数组/对象或 Base64)") } @@ -129,8 +129,9 @@ func int64Now(p *int64) int64 { return *p } -// parseSourcePayload 识别 JSON 数组 / 单对象 / Base64 / 每行一个对象。 -func parseSourcePayload(text string) []string { +// ParseSourcePayload 识别 JSON 数组 / 单对象 / Base64 / 每行一个对象, +// 返回书源 JSON 字符串列表(冒烟 CLI 复用)。 +func ParseSourcePayload(text string) []string { text = strings.TrimSpace(text) tryDecode := func(s string) []string { var arr []json.RawMessage @@ -385,7 +386,7 @@ func (s *ReaderService) Search(ctx context.Context, key string) ([]SearchBook, [ g.Go(func() error { gctxSrc, cancel := context.WithTimeout(gctx, perSourceTimeout) defer cancel() - books, err := s.searchInSource(gctxSrc, &src, key, 1) + books, err := s.searchInSource(gctxSrc, &src, nil, key, 1) mu.Lock() defer mu.Unlock() if err != nil { @@ -459,10 +460,13 @@ func mergeSearchResults(hits []searchHit, key string) []SearchBook { } // searchInSource 单源搜索(对应 WebBook.searchBook)。 -func (s *ReaderService) searchInSource(ctx context.Context, src *model.ReaderBookSource, key string, page int) ([]SearchBook, error) { - bs, err := ParseBookSource(src.RawJSON) - if err != nil { - return nil, fmt.Errorf("书源 JSON 解析失败") +func (s *ReaderService) searchInSource(ctx context.Context, src *model.ReaderBookSource, bs *BookSource, key string, page int) ([]SearchBook, error) { + if bs == nil { + var err error + bs, err = ParseBookSource(src.RawJSON) + if err != nil { + return nil, fmt.Errorf("书源 JSON 解析失败") + } } searchURL := SPtr(bs.SearchURL) if searchURL == "" { @@ -580,6 +584,10 @@ func (s *ReaderService) GetBookInfo(ctx context.Context, sourceID, sourceURL, bo if err != nil { return nil, err } + return s.getBookInfoFrom(ctx, src, bs, bookURL) +} + +func (s *ReaderService) getBookInfoFrom(ctx context.Context, src *model.ReaderBookSource, bs *BookSource, bookURL string) (*BookInfo, error) { bir := bs.RuleBookInfo if bir == nil { return nil, fmt.Errorf("书源未配置详情规则") @@ -664,6 +672,10 @@ func (s *ReaderService) GetToc(ctx context.Context, sourceID, sourceURL, bookURL if err != nil { return nil, err } + return s.getTocFrom(ctx, src, bs, bookURL, tocURL) +} + +func (s *ReaderService) getTocFrom(ctx context.Context, src *model.ReaderBookSource, bs *BookSource, bookURL, tocURL string) ([]TocChapter, error) { tr := bs.RuleToc if tr == nil || SPtr(tr.ChapterList) == "" { return nil, fmt.Errorf("书源未配置目录规则") @@ -728,6 +740,10 @@ func (s *ReaderService) GetContent(ctx context.Context, sourceID, sourceURL, boo if err != nil { return nil, err } + return s.getContentFrom(ctx, src, bs, bookURL, chapterURL) +} + +func (s *ReaderService) getContentFrom(ctx context.Context, src *model.ReaderBookSource, bs *BookSource, bookURL, chapterURL string) (*ChapterContent, error) { cr := bs.RuleContent if cr == nil || SPtr(cr.Content) == "" { return nil, fmt.Errorf("书源未配置正文规则") @@ -875,76 +891,293 @@ func (s *ReaderService) ListReplaceRules(ctx context.Context, userID string) ([] return s.repo.ListReplaceRules(ctx, userID) } -// ─── 书源调试(对应 BookSourceDebugModel 全链路) ─────────────────────────── +// ReplaceRuleInput 替换规则输入。 +type ReplaceRuleInput struct { + Name string `json:"name"` + GroupName string `json:"group"` + Pattern string `json:"pattern"` + Replacement string `json:"replacement"` + Scope string `json:"scope"` + ScopeTitle bool `json:"scope_title"` + ScopeContent bool `json:"scope_content"` + ExcludeScope string `json:"exclude_scope"` + IsEnabled bool `json:"is_enabled"` + IsRegex bool `json:"is_regex"` + TimeoutMillisecond int64 `json:"timeout_millisecond"` + Order int `json:"order"` +} -// Debug 全链路调试:搜索 → 详情 → 目录 → 正文,返回逐条日志。 -func (s *ReaderService) Debug(ctx context.Context, sourceID, key string) ([]string, error) { - src, _, err := s.loadSource(ctx, sourceID) +// CreateReplaceRule 新增替换规则。 +func (s *ReaderService) CreateReplaceRule(ctx context.Context, userID string, in ReplaceRuleInput) (*model.ReaderReplaceRule, error) { + if strings.TrimSpace(in.Pattern) == "" { + return nil, fmt.Errorf("替换规则不能为空") + } + rule := &model.ReaderReplaceRule{ + UserID: userID, + Name: in.Name, + GroupName: in.GroupName, + Pattern: in.Pattern, + Replacement: in.Replacement, + Scope: in.Scope, + ScopeTitle: in.ScopeTitle, + ScopeContent: in.ScopeContent, + ExcludeScope: in.ExcludeScope, + IsEnabled: in.IsEnabled, + IsRegex: in.IsRegex, + TimeoutMillisecond: in.TimeoutMillisecond, + Order: in.Order, + } + if err := s.repo.CreateReplaceRule(ctx, rule); err != nil { + return nil, err + } + return rule, nil +} + +// UpdateReplaceRule 更新替换规则。 +func (s *ReaderService) UpdateReplaceRule(ctx context.Context, userID, id string, in ReplaceRuleInput) error { + existing, err := s.repo.ListReplaceRules(ctx, userID) + if err != nil { + return err + } + var target *model.ReaderReplaceRule + for i := range existing { + if existing[i].ID == id { + target = &existing[i] + break + } + } + if target == nil { + return fmt.Errorf("规则不存在") + } + target.Name = in.Name + target.GroupName = in.GroupName + target.Pattern = in.Pattern + target.Replacement = in.Replacement + target.Scope = in.Scope + target.ScopeTitle = in.ScopeTitle + target.ScopeContent = in.ScopeContent + target.ExcludeScope = in.ExcludeScope + target.IsEnabled = in.IsEnabled + target.IsRegex = in.IsRegex + target.TimeoutMillisecond = in.TimeoutMillisecond + target.Order = in.Order + return s.repo.UpdateReplaceRule(ctx, target) +} + +// DeleteReplaceRule 删除替换规则。 +func (s *ReaderService) DeleteReplaceRule(ctx context.Context, userID, id string) error { + return s.repo.DeleteReplaceRule(ctx, userID, id) +} + +// ─── 书架维度正文(含用户替换净化) ───────────────────────────────────────── + +// GetContentForBook 按书架书籍 + 章节序号取正文: +// 解析书源 → 章节缓存 → 抓正文 → 书源 replaceRegex → 用户替换净化规则。 +func (s *ReaderService) GetContentForBook(ctx context.Context, userID, bookID string, chapterIndex int) (*ChapterContent, error) { + book, err := s.repo.GetBook(ctx, bookID) if err != nil { return nil, err } - var logs []string - logf := func(format string, args ...any) { - logs = append(logs, fmt.Sprintf(format, args...)) - } - logf("搜索关键词: %s", key) - books, err := s.searchInSource(ctx, src, key, 1) + chapters, err := s.repo.ListChapters(ctx, bookID) if err != nil { - logf("搜索失败: %v", err) - return logs, nil + return nil, err } + if len(chapters) == 0 { + return nil, fmt.Errorf("章节缓存为空,请先在详情页刷新目录") + } + if chapterIndex < 0 || chapterIndex >= len(chapters) { + return nil, fmt.Errorf("章节序号越界(共 %d 章)", len(chapters)) + } + ch := chapters[chapterIndex] + out, err := s.GetContent(ctx, "", book.Origin, book.BookURL, ch.URL) + if err != nil { + return nil, err + } + if out.Type == "text" { + out.Content = s.applyUserReplaceRules(ctx, userID, book.Name, out.Content) + } + return out, nil +} + +// applyUserReplaceRules 应用启用的用户替换规则(对应 legado ReplaceRule 作用链)。 +func (s *ReaderService) applyUserReplaceRules(ctx context.Context, userID, bookName, content string) string { + if content == "" { + return content + } + rules, err := s.repo.ListReplaceRules(ctx, userID) + if err != nil { + return content + } + for _, r := range rules { + if !r.IsEnabled || r.Pattern == "" || !r.ScopeContent { + continue + } + // 作用范围 / 排除范围按书名匹配(对应 legado scope / excludeScope) + if r.Scope != "" && !strings.Contains(bookName, r.Scope) { + continue + } + if r.ExcludeScope != "" && strings.Contains(bookName, r.ExcludeScope) { + continue + } + content = rule.ApplyUserReplace(content, r.Pattern, r.Replacement, r.IsRegex, r.TimeoutMillisecond) + } + return content +} + +// ─── 书源调试(对应 BookSourceDebugModel 全链路) ─────────────────────────── + +// Debug 全链路调试:搜索 → 详情 → 目录 → 正文,返回逐条日志。 +// SmokeLog 冒烟/调试日志行。 +type SmokeLog struct { + Stage string `json:"stage"` + Level string `json:"level"` // info / error + Message string `json:"message"` +} + +// SmokeChainResult 单书源全链路冒烟结果。 +type SmokeChainResult struct { + SourceID string `json:"source_id"` + SourceName string `json:"source_name"` + SourceURL string `json:"source_url"` + Type int `json:"type"` + OK bool `json:"ok"` + FailedAt string `json:"failed_at,omitempty"` // search / info / toc / content + Error string `json:"error,omitempty"` + SearchHits int `json:"search_hits"` + Chapters int `json:"chapters"` + ContentLen int `json:"content_len"` + ElapsedMS int64 `json:"elapsed_ms"` + Logs []SmokeLog `json:"logs"` +} + +// SmokeChain 对单个书源跑 搜索→详情→目录→正文 全链路, +// 返回结构化结果(书源调试接口与冒烟 CLI 共用)。 +func (s *ReaderService) SmokeChain(ctx context.Context, sourceID string, src *model.ReaderBookSource, bs *BookSource, key string) *SmokeChainResult { + res := &SmokeChainResult{Logs: []SmokeLog{}} + if src != nil { + res.SourceID = src.ID + res.SourceName = src.Name + res.SourceURL = src.SourceURL + res.Type = src.Type + } else if bs != nil { + res.SourceName = bs.BookSourceName + res.SourceURL = bs.BookSourceURL + res.Type = bs.Type() + } + start := time.Now() + logf := func(stage, level, format string, args ...any) { + res.Logs = append(res.Logs, SmokeLog{Stage: stage, Level: level, Message: fmt.Sprintf(format, args...)}) + } + fail := func(stage string, err error) *SmokeChainResult { + res.FailedAt = stage + res.Error = err.Error() + res.ElapsedMS = time.Since(start).Milliseconds() + logf(stage, "error", "%s 失败: %v", stage, err) + return res + } + + logf("search", "info", "搜索关键词: %s", key) + books, err := s.searchInSource(ctx, src, bs, key, 1) + if err != nil { + return fail("search", err) + } + res.SearchHits = len(books) if len(books) == 0 { - logf("搜索结果为空") - return logs, nil + return fail("search", fmt.Errorf("搜索结果为空")) } - logf("搜索到 %d 条结果", len(books)) + logf("search", "info", "搜索到 %d 条结果", len(books)) for i, b := range books { if i >= 3 { break } - logf("结果[%d] %s / %s", i, b.Name, b.Author) + logf("search", "info", "结果[%d] %s / %s", i, b.Name, b.Author) } first := books[0] - logf("访问详情页: %s", first.BookURL) - info, err := s.GetBookInfo(ctx, sourceID, "", first.BookURL) + + logf("info", "info", "访问详情页: %s", first.BookURL) + info, err := s.getBookInfoFrom(ctx, src, bs, first.BookURL) if err != nil { - logf("详情失败: %v", err) - return logs, nil + return fail("info", err) } - logf("书名: %s 作者: %s 最新章节: %s", info.Name, info.Author, info.LatestChapter) - logf("访问目录页: %s", info.TocURL) - chapters, err := s.GetToc(ctx, sourceID, "", first.BookURL, info.TocURL) + logf("info", "info", "书名: %s 作者: %s 最新章节: %s", info.Name, info.Author, info.LatestChapter) + + logf("toc", "info", "访问目录页: %s", info.TocURL) + chapters, err := s.getTocFrom(ctx, src, bs, first.BookURL, info.TocURL) if err != nil { - logf("目录失败: %v", err) - return logs, nil + return fail("toc", err) } - logf("共 %d 章", len(chapters)) + res.Chapters = len(chapters) + logf("toc", "info", "共 %d 章", len(chapters)) for i, c := range chapters { if i >= 3 { break } - logf("章节[%d] %s", c.Index, c.Title) + logf("toc", "info", "章节[%d] %s", c.Index, c.Title) } - // 找第一个非卷章节读正文 + for _, c := range chapters { - if c.IsVolume { + if c.IsVolume || c.URL == "" { continue } - logf("访问正文: %s", c.URL) - content, err := s.GetContent(ctx, sourceID, "", first.BookURL, c.URL) + logf("content", "info", "访问正文: %s", c.URL) + content, err := s.getContentFrom(ctx, src, bs, first.BookURL, c.URL) if err != nil { - logf("正文失败: %v", err) - return logs, nil + return fail("content", err) } + res.ContentLen = len([]rune(content.Content)) text := content.Content - if len(text) > 200 { - text = text[:200] + "..." + if len([]rune(text)) > 200 { + text = string([]rune(text)[:200]) + "..." } - logf("正文预览: %s", text) + logf("content", "info", "正文预览(%d字): %s", res.ContentLen, text) break } - logf("调试完成") - return logs, nil + if res.ContentLen == 0 { + return fail("content", fmt.Errorf("未取到正文(可能全是卷名)")) + } + res.OK = true + res.ElapsedMS = time.Since(start).Milliseconds() + logf("done", "info", "链路完成,耗时 %dms", res.ElapsedMS) + return res +} + +// SmokeSource 直接对一段书源 JSON 跑全链路(冒烟 CLI 用,不落库)。 +func (s *ReaderService) SmokeSource(ctx context.Context, raw string, key string) *SmokeChainResult { + bs, err := ParseBookSource(raw) + var res *SmokeChainResult + if err != nil { + res = &SmokeChainResult{OK: false, FailedAt: "parse", Error: err.Error(), Logs: []SmokeLog{}} + return res + } + src := &model.ReaderBookSource{ + Name: bs.BookSourceName, + GroupName: strings.TrimSpace(SPtr(bs.BookSourceGroup)), + Type: bs.Type(), + SourceURL: bs.BookSourceURL, + RawJSON: raw, + Header: SPtr(bs.Header), + Enabled: true, + CustomOrder: IPtr(bs.CustomOrder), + } + return s.SmokeChain(ctx, "", src, bs, key) +} + +// Debug 书源调试接口:返回逐条日志字符串(前端展示用)。 +func (s *ReaderService) Debug(ctx context.Context, sourceID, key string) ([]string, error) { + src, bs, err := s.loadSource(ctx, sourceID) + if err != nil { + return nil, err + } + res := s.SmokeChain(ctx, src.ID, src, bs, key) + out := make([]string, 0, len(res.Logs)) + for _, l := range res.Logs { + prefix := "[info]" + if l.Level == "error" { + prefix = "[错误]" + } + out = append(out, fmt.Sprintf("%s %s", prefix, l.Message)) + } + return out, nil } // loadSource 加载书源记录与解析结构。 diff --git a/internal/service/reader/reader_test.go b/internal/service/reader/reader_test.go index 12a5176..e810b54 100644 --- a/internal/service/reader/reader_test.go +++ b/internal/service/reader/reader_test.go @@ -218,17 +218,17 @@ func TestEndToEndSourceChain(t *testing.T) { func TestParseSourcePayload(t *testing.T) { // 数组 arr := `[{"bookSourceUrl":"http://a.com","bookSourceName":"A"},{"bookSourceUrl":"http://b.com","bookSourceName":"B"}]` - if got := parseSourcePayload(arr); len(got) != 2 { + if got := ParseSourcePayload(arr); len(got) != 2 { t.Fatalf("array payload = %d", len(got)) } // 单对象 single := `{"bookSourceUrl":"http://a.com","bookSourceName":"A"}` - if got := parseSourcePayload(single); len(got) != 1 { + if got := ParseSourcePayload(single); len(got) != 1 { t.Fatalf("single payload = %d", len(got)) } // Base64 b64 := base64StdEncode(single) - if got := parseSourcePayload(b64); len(got) != 1 { + if got := ParseSourcePayload(b64); len(got) != 1 { t.Fatalf("base64 payload = %d", len(got)) } } diff --git a/internal/service/reader/rule/regexan.go b/internal/service/reader/rule/regexan.go index cb304a4..9492b67 100644 --- a/internal/service/reader/rule/regexan.go +++ b/internal/service/reader/rule/regexan.go @@ -2,6 +2,7 @@ package rule import ( "strings" + "time" "github.com/dlclark/regexp2" ) @@ -105,6 +106,32 @@ func regexReplaceAll(pattern, result, replacement string) string { return out } +// ApplyUserReplace 应用一条用户替换净化规则(对应 legado ReplaceRule)。 +// isRegex=false 按字面替换;isRegex=true 用 Java 正则语义并带匹配超时 +// (防灾难性回溯挂死服务),编译失败回退字面替换。 +func ApplyUserReplace(content, pattern, replacement string, isRegex bool, timeoutMS int64) string { + if pattern == "" { + return content + } + if !isRegex { + return strings.ReplaceAll(content, pattern, replacement) + } + re, err := regexp2.Compile(pattern, regexp2.None) + if err != nil { + return strings.ReplaceAll(content, pattern, replacement) + } + if timeoutMS > 0 { + re.MatchTimeout = time.Duration(timeoutMS) * time.Millisecond + } else { + re.MatchTimeout = 3 * time.Second + } + out, err := re.Replace(content, replacement, 0, -1) + if err != nil { + return content + } + return out +} + // ApplyReplaceRegexString 应用书源 replaceRegex 字符串("##pattern##replace[##x]" 格式), // 对应 ContentRule.replaceRegex 的处理。 func ApplyReplaceRegexString(content, replaceRegex string) string { diff --git a/web/src/api/reader.ts b/web/src/api/reader.ts index 9c01c67..2bd9a2d 100644 --- a/web/src/api/reader.ts +++ b/web/src/api/reader.ts @@ -152,8 +152,32 @@ export const readerAPI = { listChapters: (id: string) => api.get<{ chapters: ReaderChapter[] }>(`/reader/books/${id}/chapters`).then((r) => r.data.chapters), saveChapters: (id: string, chapters: ReaderChapter[]) => api.post(`/reader/books/${id}/chapters`, { chapters }), + // 书架维度正文(服务端已应用书源 replaceRegex 与用户替换净化规则) + bookContent: (id: string, chapter: number) => + api + .get(`/reader/books/${id}/content`, { params: { chapter }, timeout: LONG_REQUEST_TIMEOUT }) + .then((r) => r.data), - // ── 替换规则 ── + // ── 替换净化规则 ── listReplaceRules: () => api.get<{ rules: ReaderReplaceRule[] }>('/reader/replace-rules').then((r) => r.data.rules), + createReplaceRule: (body: ReplaceRuleInput) => + api.post('/reader/replace-rules', body).then((r) => r.data), + updateReplaceRule: (id: string, body: ReplaceRuleInput) => api.patch(`/reader/replace-rules/${id}`, body), + deleteReplaceRule: (id: string) => api.delete(`/reader/replace-rules/${id}`), +} + +export interface ReplaceRuleInput { + name: string + group: string + pattern: string + replacement: string + scope: string + scope_title: boolean + scope_content: boolean + exclude_scope: string + is_enabled: boolean + is_regex: boolean + timeout_millisecond: number + order: number } diff --git a/web/src/pages/reader/ReaderReplacePage.tsx b/web/src/pages/reader/ReaderReplacePage.tsx new file mode 100644 index 0000000..8305323 --- /dev/null +++ b/web/src/pages/reader/ReaderReplacePage.tsx @@ -0,0 +1,215 @@ +import { useEffect, useState } from 'react' +import { useNavigate } from 'react-router-dom' +import toast from 'react-hot-toast' +import { ArrowLeft, Loader2, Plus, Trash2 } from 'lucide-react' + +import { readerAPI, type ReaderReplaceRule, type ReplaceRuleInput } from '../../api/reader' + +// 替换净化规则页(仿 legado ReplaceRuleActivity:列表 + 启停 + 编辑)。 + +const emptyInput: ReplaceRuleInput = { + name: '', + group: '', + pattern: '', + replacement: '', + scope: '', + scope_title: false, + scope_content: true, + exclude_scope: '', + is_enabled: true, + is_regex: true, + timeout_millisecond: 3000, + order: 0, +} + +export default function ReaderReplacePage() { + const navigate = useNavigate() + const [rules, setRules] = useState(null) + const [showForm, setShowForm] = useState(false) + const [form, setForm] = useState(emptyInput) + const [saving, setSaving] = useState(false) + + const load = () => { + readerAPI + .listReplaceRules() + .then(setRules) + .catch(() => toast.error('加载替换规则失败')) + } + // eslint-disable-next-line react-hooks/exhaustive-deps + useEffect(load, []) + + const save = async () => { + if (!form.pattern.trim()) { + toast.error('替换规则不能为空') + return + } + setSaving(true) + try { + await readerAPI.createReplaceRule(form) + toast.success('已添加') + setForm(emptyInput) + setShowForm(false) + load() + } catch (e) { + toast.error((e as { response?: { data?: { error?: string } } })?.response?.data?.error ?? '保存失败') + } finally { + setSaving(false) + } + } + + const toggle = async (r: ReaderReplaceRule) => { + try { + await readerAPI.updateReplaceRule(r.id, { + name: r.name, + group: r.group, + pattern: r.pattern, + replacement: r.replacement, + scope: r.scope, + scope_title: r.scope_title, + scope_content: r.scope_content, + exclude_scope: r.exclude_scope, + is_enabled: !r.is_enabled, + is_regex: r.is_regex, + timeout_millisecond: 3000, + order: r.order, + }) + setRules((prev) => prev?.map((x) => (x.id === r.id ? { ...x, is_enabled: !x.is_enabled } : x)) ?? null) + } catch { + toast.error('更新失败') + } + } + + const remove = async (r: ReaderReplaceRule) => { + if (!window.confirm(`删除规则「${r.name || r.pattern}」?`)) return + try { + await readerAPI.deleteReplaceRule(r.id) + setRules((prev) => prev?.filter((x) => x.id !== r.id) ?? null) + } catch { + toast.error('删除失败') + } + } + + return ( +
+
+ +

替换净化

+ +
+ + {showForm && ( +
+
+ setForm({ ...form, name: e.target.value })} + placeholder="规则名(可选)" + className="rounded-xl border border-[var(--app-border)] bg-[var(--app-panel-soft)] px-3 py-2 text-xs text-[var(--app-text)] outline-none" + /> + setForm({ ...form, group: e.target.value })} + placeholder="分组(可选)" + className="rounded-xl border border-[var(--app-border)] bg-[var(--app-panel-soft)] px-3 py-2 text-xs text-[var(--app-text)] outline-none" + /> +
+