feat(reader): P2 收尾——冒烟 CLI、书架正文链路、替换净化规则

- SmokeChain 结构化全链路冒烟(搜索→详情→目录→正文,分阶段失败定位),
  SmokeSource 支持不落库直测;Debug 接口改为结构化日志输出
- cmd/reader-smoke:批量书源冒烟 CLI(-file/-url 导入、并发、
  各阶段通过率汇总、-json 完整报告)
- GetContentForBook:书架维度正文(章节缓存 + 书源 replaceRegex +
  用户替换净化规则),前端阅读器切换到该接口
- 替换净化规则 CRUD(后端 + /reader/replace 前端页),用户规则
  按 scope/excludeScope 书名过滤、正则带匹配超时防回溯挂死
- rule.ApplyUserReplace 导出(regexp2 MatchTimeout)
This commit is contained in:
truewhile
2026-09-30 16:55:52 +08:00
parent 5976a8d310
commit 9cc21a47c5
11 changed files with 828 additions and 65 deletions
+197
View File
@@ -0,0 +1,197 @@
// reader-smoke 是阅读书源兼容性冒烟工具:
// 对批量书源逐个跑「搜索 → 详情 → 目录 → 正文」全链路,输出兼容率报告。
//
// 用法:
//
// go run ./cmd/reader-smoke -file sources.json -key 斗破苍穹 -c 8
// go run ./cmd/reader-smoke -url https://example.com/sources.json -json > report.json
package main
import (
"context"
"encoding/json"
"flag"
"fmt"
"net/http"
"os"
"strings"
"sync"
"time"
"go.uber.org/zap"
"github.com/truewhile/MeBox/internal/config"
"github.com/truewhile/MeBox/internal/helper"
"github.com/truewhile/MeBox/internal/repository"
"github.com/truewhile/MeBox/internal/service/reader"
)
func main() {
file := flag.String("file", "", "书源文件路径(JSON 数组/对象/Base64/每行一个)")
urlFlag := flag.String("url", "", "书源网络地址(与 -file 二选一)")
key := flag.String("key", "斗破苍穹", "搜索关键词")
concurrency := flag.Int("c", 4, "并发数")
timeout := flag.Int("timeout", 90, "单源全链路超时(秒)")
jsonOut := flag.Bool("json", false, "输出完整 JSON 报告(追加在汇总后)")
flag.Parse()
payload := ""
switch {
case *file != "":
data, err := os.ReadFile(*file)
if err != nil {
fatal("读取文件失败: %v", err)
}
payload = string(data)
case *urlFlag != "":
client := helper.NewSiteHTTPClient(30, true)
req, err := http.NewRequest("GET", *urlFlag, nil)
if err != nil {
fatal("构造请求失败: %v", err)
}
for k, v := range helper.HTTPHeaderPresets() {
req.Header.Set(k, v)
}
resp, err := client.Do(req)
if err != nil {
fatal("拉取书源失败: %v", err)
}
defer resp.Body.Close()
var sb strings.Builder
buf := make([]byte, 32*1024)
for {
n, err := resp.Body.Read(buf)
sb.Write(buf[:n])
if err != nil {
break
}
}
payload = sb.String()
default:
fatal("需要 -file 或 -url 指定书源来源")
}
sources := reader.ParseSourcePayload(payload)
if len(sources) == 0 {
fatal("未从输入中识别到书源")
}
svc := reader.NewReaderService(&config.Config{}, zap.NewNop(), &repository.Container{})
ctx := context.Background()
results := make([]*reader.SmokeChainResult, len(sources))
sem := make(chan struct{}, max(1, *concurrency))
var wg sync.WaitGroup
for i, raw := range sources {
wg.Add(1)
sem <- struct{}{}
go func(i int, raw string) {
defer wg.Done()
defer func() { <-sem }()
ctxSrc, cancel := context.WithTimeout(ctx, time.Duration(*timeout)*time.Second)
defer cancel()
res := svc.SmokeSource(ctxSrc, raw, *key)
results[i] = res
status := "✓"
if !res.OK {
status = "✗"
}
fmt.Fprintf(os.Stderr, "%s %-24s [%s] hits=%d chapters=%d content=%d %s\n",
status, truncate(res.SourceName, 24), stageCN(res), res.SearchHits, res.Chapters, res.ContentLen, res.Error)
}(i, raw)
}
wg.Wait()
// 汇总
var searchOK, infoOK, tocOK, contentOK, allOK int
failedAt := map[string]int{}
for _, r := range results {
if r == nil {
continue
}
switch r.FailedAt {
case "":
allOK++
searchOK++
infoOK++
tocOK++
contentOK++
case "search":
failedAt["search"]++
case "info":
searchOK++
failedAt["info"]++
case "toc":
searchOK++
infoOK++
failedAt["toc"]++
case "content":
searchOK++
infoOK++
tocOK++
failedAt["content"]++
}
}
n := len(results)
pct := func(v int) string {
if n == 0 {
return "0%"
}
return fmt.Sprintf("%.1f%%", float64(v)/float64(n)*100)
}
fmt.Printf("\n==== 冒烟报告 ====\n")
fmt.Printf("书源总数: %d 关键词: %s\n", n, *key)
fmt.Printf("搜索通过: %d (%s)\n", searchOK, pct(searchOK))
fmt.Printf("详情通过: %d (%s)\n", infoOK, pct(infoOK))
fmt.Printf("目录通过: %d (%s)\n", tocOK, pct(tocOK))
fmt.Printf("正文通过: %d (%s)\n", contentOK, pct(contentOK))
fmt.Printf("全链路通过: %d (%s)\n", allOK, pct(allOK))
for _, stage := range []string{"search", "info", "toc", "content"} {
if failedAt[stage] > 0 {
fmt.Printf(" 失败于 %s: %d\n", stageCN(&reader.SmokeChainResult{FailedAt: stage}), failedAt[stage])
}
}
if *jsonOut {
out, err := json.MarshalIndent(results, "", " ")
if err != nil {
fatal("序列化报告失败: %v", err)
}
fmt.Println(string(out))
}
}
func stageCN(r *reader.SmokeChainResult) string {
switch r.FailedAt {
case "":
return "完成"
case "search":
return "搜索"
case "info":
return "详情"
case "toc":
return "目录"
case "content":
return "正文"
case "parse":
return "解析"
default:
return r.FailedAt
}
}
func truncate(s string, n int) string {
rs := []rune(strings.TrimSpace(s))
if len(rs) <= n {
if len(rs) == 0 {
return "(未命名)"
}
return string(rs)
}
return string(rs[:n]) + "…"
}
func fatal(format string, args ...any) {
fmt.Fprintf(os.Stderr, "reader-smoke: "+format+"\n", args...)
os.Exit(1)
}
+1 -1
View File
@@ -119,7 +119,7 @@ web/src/
|---|---|---|
| P0 引擎地基 ✅ | 规则引擎核心(四分析器 + 规则拆分/组合/变量)+ AnalyzeUrl v1(GET/POST/charset/headers/变量/页码模式)+ 表结构 + 书源导入/管理 API + 搜索/详情/目录/正文/书架/进度/调试 API | 已完成:`internal/service/reader/rule/`(规则引擎,~2800 行,对齐 AnalyzeRule/AnalyzeByJSoup/AnalyzeByJSonPath/AnalyzeByXPath/AnalyzeByRegex/AnalyzeUrl/RuleAnalyzer)+ 服务层 + `/api/reader/*` 路由 + 单测/端到端测试全绿 |
| P1 文本源全链路 + 首页切换 | 搜索聚合(WS 进度)/详情/目录/正文(nextContentUrl 合并、缓存)+ 前端首页切换、书架、搜索、详情、文本阅读器 v1(阅读器样式仿 legado:9 宫格点击、主题、翻页动画) | 用纯规则型文本源完成「搜书→加入→阅读」全流程 |
| P2 JS 与兼容率爬坡 | goja 接入 + `java.*` 桥分批实现 + 加解密族 + URL 完整选项 + replaceRegex + 替换规则管理 + 书源调试页(仿 legado 逐条日志流式输出)+ 冒烟指标报告 | 主流公开文本源通过率显著提升,形成回归基线 |
| P2 JS 与兼容率爬坡 ✅ | goja 接入 + `java.*` 桥(网络/编解码/摘要/对称加密全家桶/规则回调,函数名对齐 JsExtensions)+ URL 规则 JS(analyzeJs/{{}}/js/bodyJs)+ cookie jar + 用户替换净化规则(含正则超时保护)+ 替换净化页 + 结构化冒烟链路(SmokeChain)+ `cmd/reader-smoke` 冒烟 CLI | JS 源可用;冒烟 CLI 跑公开书源集出各阶段通过率报告 |
| P3 音频源 | 播放列表解析、音频代理(带 UA/Referer)、音频播放器页(仿 ReadAloudDialog 布局:上一章/播放/下一章/定时/倍速)、进度记忆 | 音频源可听 |
| P4 漫画/图片源 | 图片列表解析(含翻页)、图片代理接入磁盘缓存、漫画阅读器双模式(MangaMenu:顶栏+底部胶囊)、预加载 | 漫画源可看 |
| P5 体验完善 | 换源(ChangeBookSourceDialog 四档排序)、追更(定时刷新目录 + 缓存清理)、发现页(exploreUrl 标签条)、阅读器高级设置(页眉页脚提示、点击区域自定义)、书源编辑器六 Tab、备份导出;可选:本地 TXT/EPUB | 完整体验 |
+67
View File
@@ -3,6 +3,7 @@ package handler
import (
"net/http"
"strconv"
"github.com/gin-gonic/gin"
@@ -36,9 +37,13 @@ func registerReaderRoutes(authed *gin.RouterGroup, svc *service.Container) {
g.PUT("/books/:id/progress", readerSaveProgressHandler(svc))
g.GET("/books/:id/chapters", readerListChaptersHandler(svc))
g.POST("/books/:id/chapters", readerReplaceChaptersHandler(svc))
g.GET("/books/:id/content", readerBookContentHandler(svc))
// 替换净化规则
g.GET("/replace-rules", readerListReplaceRulesHandler(svc))
g.POST("/replace-rules", readerCreateReplaceRuleHandler(svc))
g.PATCH("/replace-rules/:id", readerUpdateReplaceRuleHandler(svc))
g.DELETE("/replace-rules/:id", readerDeleteReplaceRuleHandler(svc))
}
func readerListSourcesHandler(svc *service.Container) gin.HandlerFunc {
@@ -279,3 +284,65 @@ func readerListReplaceRulesHandler(svc *service.Container) gin.HandlerFunc {
c.JSON(http.StatusOK, gin.H{"rules": rules})
}
}
func readerCreateReplaceRuleHandler(svc *service.Container) gin.HandlerFunc {
var body reader.ReplaceRuleInput
return func(c *gin.Context) {
if err := c.ShouldBindJSON(&body); err != nil {
c.JSON(http.StatusBadRequest, gin.H{"error": err.Error()})
return
}
userID := c.GetString(middleware.CtxUserID)
rule, err := svc.Reader.CreateReplaceRule(c.Request.Context(), userID, body)
if err != nil {
c.JSON(http.StatusBadRequest, gin.H{"error": err.Error()})
return
}
c.JSON(http.StatusOK, rule)
}
}
func readerUpdateReplaceRuleHandler(svc *service.Container) gin.HandlerFunc {
var body reader.ReplaceRuleInput
return func(c *gin.Context) {
if err := c.ShouldBindJSON(&body); err != nil {
c.JSON(http.StatusBadRequest, gin.H{"error": err.Error()})
return
}
userID := c.GetString(middleware.CtxUserID)
if err := svc.Reader.UpdateReplaceRule(c.Request.Context(), userID, c.Param("id"), body); err != nil {
c.JSON(http.StatusBadRequest, gin.H{"error": err.Error()})
return
}
c.JSON(http.StatusOK, gin.H{"ok": true})
}
}
func readerDeleteReplaceRuleHandler(svc *service.Container) gin.HandlerFunc {
return func(c *gin.Context) {
userID := c.GetString(middleware.CtxUserID)
if err := svc.Reader.DeleteReplaceRule(c.Request.Context(), userID, c.Param("id")); err != nil {
c.JSON(http.StatusInternalServerError, gin.H{"error": err.Error()})
return
}
c.JSON(http.StatusOK, gin.H{"ok": true})
}
}
// readerBookContentHandler 书架维度正文(服务端应用替换净化规则)。
func readerBookContentHandler(svc *service.Container) gin.HandlerFunc {
return func(c *gin.Context) {
chapter, err := strconv.Atoi(c.DefaultQuery("chapter", "0"))
if err != nil {
c.JSON(http.StatusBadRequest, gin.H{"error": "chapter 参数需为整数"})
return
}
userID := c.GetString(middleware.CtxUserID)
content, err := svc.Reader.GetContentForBook(c.Request.Context(), userID, c.Param("id"), chapter)
if err != nil {
c.JSON(http.StatusBadRequest, gin.H{"error": err.Error()})
return
}
c.JSON(http.StatusOK, content)
}
}
+279 -46
View File
@@ -65,7 +65,7 @@ func (s *ReaderService) ImportSources(ctx context.Context, text string) (int, er
}
text = body
}
sources := parseSourcePayload(text)
sources := ParseSourcePayload(text)
if len(sources) == 0 {
return 0, fmt.Errorf("未识别到有效书源(支持 JSON 数组/对象或 Base64)")
}
@@ -129,8 +129,9 @@ func int64Now(p *int64) int64 {
return *p
}
// parseSourcePayload 识别 JSON 数组 / 单对象 / Base64 / 每行一个对象。
func parseSourcePayload(text string) []string {
// ParseSourcePayload 识别 JSON 数组 / 单对象 / Base64 / 每行一个对象,
// 返回书源 JSON 字符串列表(冒烟 CLI 复用)。
func ParseSourcePayload(text string) []string {
text = strings.TrimSpace(text)
tryDecode := func(s string) []string {
var arr []json.RawMessage
@@ -385,7 +386,7 @@ func (s *ReaderService) Search(ctx context.Context, key string) ([]SearchBook, [
g.Go(func() error {
gctxSrc, cancel := context.WithTimeout(gctx, perSourceTimeout)
defer cancel()
books, err := s.searchInSource(gctxSrc, &src, key, 1)
books, err := s.searchInSource(gctxSrc, &src, nil, key, 1)
mu.Lock()
defer mu.Unlock()
if err != nil {
@@ -459,10 +460,13 @@ func mergeSearchResults(hits []searchHit, key string) []SearchBook {
}
// searchInSource 单源搜索(对应 WebBook.searchBook)。
func (s *ReaderService) searchInSource(ctx context.Context, src *model.ReaderBookSource, key string, page int) ([]SearchBook, error) {
bs, err := ParseBookSource(src.RawJSON)
if err != nil {
return nil, fmt.Errorf("书源 JSON 解析失败")
func (s *ReaderService) searchInSource(ctx context.Context, src *model.ReaderBookSource, bs *BookSource, key string, page int) ([]SearchBook, error) {
if bs == nil {
var err error
bs, err = ParseBookSource(src.RawJSON)
if err != nil {
return nil, fmt.Errorf("书源 JSON 解析失败")
}
}
searchURL := SPtr(bs.SearchURL)
if searchURL == "" {
@@ -580,6 +584,10 @@ func (s *ReaderService) GetBookInfo(ctx context.Context, sourceID, sourceURL, bo
if err != nil {
return nil, err
}
return s.getBookInfoFrom(ctx, src, bs, bookURL)
}
func (s *ReaderService) getBookInfoFrom(ctx context.Context, src *model.ReaderBookSource, bs *BookSource, bookURL string) (*BookInfo, error) {
bir := bs.RuleBookInfo
if bir == nil {
return nil, fmt.Errorf("书源未配置详情规则")
@@ -664,6 +672,10 @@ func (s *ReaderService) GetToc(ctx context.Context, sourceID, sourceURL, bookURL
if err != nil {
return nil, err
}
return s.getTocFrom(ctx, src, bs, bookURL, tocURL)
}
func (s *ReaderService) getTocFrom(ctx context.Context, src *model.ReaderBookSource, bs *BookSource, bookURL, tocURL string) ([]TocChapter, error) {
tr := bs.RuleToc
if tr == nil || SPtr(tr.ChapterList) == "" {
return nil, fmt.Errorf("书源未配置目录规则")
@@ -728,6 +740,10 @@ func (s *ReaderService) GetContent(ctx context.Context, sourceID, sourceURL, boo
if err != nil {
return nil, err
}
return s.getContentFrom(ctx, src, bs, bookURL, chapterURL)
}
func (s *ReaderService) getContentFrom(ctx context.Context, src *model.ReaderBookSource, bs *BookSource, bookURL, chapterURL string) (*ChapterContent, error) {
cr := bs.RuleContent
if cr == nil || SPtr(cr.Content) == "" {
return nil, fmt.Errorf("书源未配置正文规则")
@@ -875,76 +891,293 @@ func (s *ReaderService) ListReplaceRules(ctx context.Context, userID string) ([]
return s.repo.ListReplaceRules(ctx, userID)
}
// ─── 书源调试(对应 BookSourceDebugModel 全链路) ───────────────────────────
// ReplaceRuleInput 替换规则输入。
type ReplaceRuleInput struct {
Name string `json:"name"`
GroupName string `json:"group"`
Pattern string `json:"pattern"`
Replacement string `json:"replacement"`
Scope string `json:"scope"`
ScopeTitle bool `json:"scope_title"`
ScopeContent bool `json:"scope_content"`
ExcludeScope string `json:"exclude_scope"`
IsEnabled bool `json:"is_enabled"`
IsRegex bool `json:"is_regex"`
TimeoutMillisecond int64 `json:"timeout_millisecond"`
Order int `json:"order"`
}
// Debug 全链路调试:搜索 → 详情 → 目录 → 正文,返回逐条日志。
func (s *ReaderService) Debug(ctx context.Context, sourceID, key string) ([]string, error) {
src, _, err := s.loadSource(ctx, sourceID)
// CreateReplaceRule 新增替换规则。
func (s *ReaderService) CreateReplaceRule(ctx context.Context, userID string, in ReplaceRuleInput) (*model.ReaderReplaceRule, error) {
if strings.TrimSpace(in.Pattern) == "" {
return nil, fmt.Errorf("替换规则不能为空")
}
rule := &model.ReaderReplaceRule{
UserID: userID,
Name: in.Name,
GroupName: in.GroupName,
Pattern: in.Pattern,
Replacement: in.Replacement,
Scope: in.Scope,
ScopeTitle: in.ScopeTitle,
ScopeContent: in.ScopeContent,
ExcludeScope: in.ExcludeScope,
IsEnabled: in.IsEnabled,
IsRegex: in.IsRegex,
TimeoutMillisecond: in.TimeoutMillisecond,
Order: in.Order,
}
if err := s.repo.CreateReplaceRule(ctx, rule); err != nil {
return nil, err
}
return rule, nil
}
// UpdateReplaceRule 更新替换规则。
func (s *ReaderService) UpdateReplaceRule(ctx context.Context, userID, id string, in ReplaceRuleInput) error {
existing, err := s.repo.ListReplaceRules(ctx, userID)
if err != nil {
return err
}
var target *model.ReaderReplaceRule
for i := range existing {
if existing[i].ID == id {
target = &existing[i]
break
}
}
if target == nil {
return fmt.Errorf("规则不存在")
}
target.Name = in.Name
target.GroupName = in.GroupName
target.Pattern = in.Pattern
target.Replacement = in.Replacement
target.Scope = in.Scope
target.ScopeTitle = in.ScopeTitle
target.ScopeContent = in.ScopeContent
target.ExcludeScope = in.ExcludeScope
target.IsEnabled = in.IsEnabled
target.IsRegex = in.IsRegex
target.TimeoutMillisecond = in.TimeoutMillisecond
target.Order = in.Order
return s.repo.UpdateReplaceRule(ctx, target)
}
// DeleteReplaceRule 删除替换规则。
func (s *ReaderService) DeleteReplaceRule(ctx context.Context, userID, id string) error {
return s.repo.DeleteReplaceRule(ctx, userID, id)
}
// ─── 书架维度正文(含用户替换净化) ─────────────────────────────────────────
// GetContentForBook 按书架书籍 + 章节序号取正文:
// 解析书源 → 章节缓存 → 抓正文 → 书源 replaceRegex → 用户替换净化规则。
func (s *ReaderService) GetContentForBook(ctx context.Context, userID, bookID string, chapterIndex int) (*ChapterContent, error) {
book, err := s.repo.GetBook(ctx, bookID)
if err != nil {
return nil, err
}
var logs []string
logf := func(format string, args ...any) {
logs = append(logs, fmt.Sprintf(format, args...))
}
logf("搜索关键词: %s", key)
books, err := s.searchInSource(ctx, src, key, 1)
chapters, err := s.repo.ListChapters(ctx, bookID)
if err != nil {
logf("搜索失败: %v", err)
return logs, nil
return nil, err
}
if len(chapters) == 0 {
return nil, fmt.Errorf("章节缓存为空,请先在详情页刷新目录")
}
if chapterIndex < 0 || chapterIndex >= len(chapters) {
return nil, fmt.Errorf("章节序号越界(共 %d 章)", len(chapters))
}
ch := chapters[chapterIndex]
out, err := s.GetContent(ctx, "", book.Origin, book.BookURL, ch.URL)
if err != nil {
return nil, err
}
if out.Type == "text" {
out.Content = s.applyUserReplaceRules(ctx, userID, book.Name, out.Content)
}
return out, nil
}
// applyUserReplaceRules 应用启用的用户替换规则(对应 legado ReplaceRule 作用链)。
func (s *ReaderService) applyUserReplaceRules(ctx context.Context, userID, bookName, content string) string {
if content == "" {
return content
}
rules, err := s.repo.ListReplaceRules(ctx, userID)
if err != nil {
return content
}
for _, r := range rules {
if !r.IsEnabled || r.Pattern == "" || !r.ScopeContent {
continue
}
// 作用范围 / 排除范围按书名匹配(对应 legado scope / excludeScope)
if r.Scope != "" && !strings.Contains(bookName, r.Scope) {
continue
}
if r.ExcludeScope != "" && strings.Contains(bookName, r.ExcludeScope) {
continue
}
content = rule.ApplyUserReplace(content, r.Pattern, r.Replacement, r.IsRegex, r.TimeoutMillisecond)
}
return content
}
// ─── 书源调试(对应 BookSourceDebugModel 全链路) ───────────────────────────
// Debug 全链路调试:搜索 → 详情 → 目录 → 正文,返回逐条日志。
// SmokeLog 冒烟/调试日志行。
type SmokeLog struct {
Stage string `json:"stage"`
Level string `json:"level"` // info / error
Message string `json:"message"`
}
// SmokeChainResult 单书源全链路冒烟结果。
type SmokeChainResult struct {
SourceID string `json:"source_id"`
SourceName string `json:"source_name"`
SourceURL string `json:"source_url"`
Type int `json:"type"`
OK bool `json:"ok"`
FailedAt string `json:"failed_at,omitempty"` // search / info / toc / content
Error string `json:"error,omitempty"`
SearchHits int `json:"search_hits"`
Chapters int `json:"chapters"`
ContentLen int `json:"content_len"`
ElapsedMS int64 `json:"elapsed_ms"`
Logs []SmokeLog `json:"logs"`
}
// SmokeChain 对单个书源跑 搜索→详情→目录→正文 全链路,
// 返回结构化结果(书源调试接口与冒烟 CLI 共用)。
func (s *ReaderService) SmokeChain(ctx context.Context, sourceID string, src *model.ReaderBookSource, bs *BookSource, key string) *SmokeChainResult {
res := &SmokeChainResult{Logs: []SmokeLog{}}
if src != nil {
res.SourceID = src.ID
res.SourceName = src.Name
res.SourceURL = src.SourceURL
res.Type = src.Type
} else if bs != nil {
res.SourceName = bs.BookSourceName
res.SourceURL = bs.BookSourceURL
res.Type = bs.Type()
}
start := time.Now()
logf := func(stage, level, format string, args ...any) {
res.Logs = append(res.Logs, SmokeLog{Stage: stage, Level: level, Message: fmt.Sprintf(format, args...)})
}
fail := func(stage string, err error) *SmokeChainResult {
res.FailedAt = stage
res.Error = err.Error()
res.ElapsedMS = time.Since(start).Milliseconds()
logf(stage, "error", "%s 失败: %v", stage, err)
return res
}
logf("search", "info", "搜索关键词: %s", key)
books, err := s.searchInSource(ctx, src, bs, key, 1)
if err != nil {
return fail("search", err)
}
res.SearchHits = len(books)
if len(books) == 0 {
logf("搜索结果为空")
return logs, nil
return fail("search", fmt.Errorf("搜索结果为空"))
}
logf("搜索到 %d 条结果", len(books))
logf("search", "info", "搜索到 %d 条结果", len(books))
for i, b := range books {
if i >= 3 {
break
}
logf("结果[%d] %s / %s", i, b.Name, b.Author)
logf("search", "info", "结果[%d] %s / %s", i, b.Name, b.Author)
}
first := books[0]
logf("访问详情页: %s", first.BookURL)
info, err := s.GetBookInfo(ctx, sourceID, "", first.BookURL)
logf("info", "info", "访问详情页: %s", first.BookURL)
info, err := s.getBookInfoFrom(ctx, src, bs, first.BookURL)
if err != nil {
logf("详情失败: %v", err)
return logs, nil
return fail("info", err)
}
logf("书名: %s 作者: %s 最新章节: %s", info.Name, info.Author, info.LatestChapter)
logf("访问目录页: %s", info.TocURL)
chapters, err := s.GetToc(ctx, sourceID, "", first.BookURL, info.TocURL)
logf("info", "info", "书名: %s 作者: %s 最新章节: %s", info.Name, info.Author, info.LatestChapter)
logf("toc", "info", "访问目录页: %s", info.TocURL)
chapters, err := s.getTocFrom(ctx, src, bs, first.BookURL, info.TocURL)
if err != nil {
logf("目录失败: %v", err)
return logs, nil
return fail("toc", err)
}
logf("共 %d 章", len(chapters))
res.Chapters = len(chapters)
logf("toc", "info", "共 %d 章", len(chapters))
for i, c := range chapters {
if i >= 3 {
break
}
logf("章节[%d] %s", c.Index, c.Title)
logf("toc", "info", "章节[%d] %s", c.Index, c.Title)
}
// 找第一个非卷章节读正文
for _, c := range chapters {
if c.IsVolume {
if c.IsVolume || c.URL == "" {
continue
}
logf("访问正文: %s", c.URL)
content, err := s.GetContent(ctx, sourceID, "", first.BookURL, c.URL)
logf("content", "info", "访问正文: %s", c.URL)
content, err := s.getContentFrom(ctx, src, bs, first.BookURL, c.URL)
if err != nil {
logf("正文失败: %v", err)
return logs, nil
return fail("content", err)
}
res.ContentLen = len([]rune(content.Content))
text := content.Content
if len(text) > 200 {
text = text[:200] + "..."
if len([]rune(text)) > 200 {
text = string([]rune(text)[:200]) + "..."
}
logf("正文预览: %s", text)
logf("content", "info", "正文预览(%d字): %s", res.ContentLen, text)
break
}
logf("调试完成")
return logs, nil
if res.ContentLen == 0 {
return fail("content", fmt.Errorf("未取到正文(可能全是卷名)"))
}
res.OK = true
res.ElapsedMS = time.Since(start).Milliseconds()
logf("done", "info", "链路完成,耗时 %dms", res.ElapsedMS)
return res
}
// SmokeSource 直接对一段书源 JSON 跑全链路(冒烟 CLI 用,不落库)。
func (s *ReaderService) SmokeSource(ctx context.Context, raw string, key string) *SmokeChainResult {
bs, err := ParseBookSource(raw)
var res *SmokeChainResult
if err != nil {
res = &SmokeChainResult{OK: false, FailedAt: "parse", Error: err.Error(), Logs: []SmokeLog{}}
return res
}
src := &model.ReaderBookSource{
Name: bs.BookSourceName,
GroupName: strings.TrimSpace(SPtr(bs.BookSourceGroup)),
Type: bs.Type(),
SourceURL: bs.BookSourceURL,
RawJSON: raw,
Header: SPtr(bs.Header),
Enabled: true,
CustomOrder: IPtr(bs.CustomOrder),
}
return s.SmokeChain(ctx, "", src, bs, key)
}
// Debug 书源调试接口:返回逐条日志字符串(前端展示用)。
func (s *ReaderService) Debug(ctx context.Context, sourceID, key string) ([]string, error) {
src, bs, err := s.loadSource(ctx, sourceID)
if err != nil {
return nil, err
}
res := s.SmokeChain(ctx, src.ID, src, bs, key)
out := make([]string, 0, len(res.Logs))
for _, l := range res.Logs {
prefix := "[info]"
if l.Level == "error" {
prefix = "[错误]"
}
out = append(out, fmt.Sprintf("%s %s", prefix, l.Message))
}
return out, nil
}
// loadSource 加载书源记录与解析结构。
+3 -3
View File
@@ -218,17 +218,17 @@ func TestEndToEndSourceChain(t *testing.T) {
func TestParseSourcePayload(t *testing.T) {
// 数组
arr := `[{"bookSourceUrl":"http://a.com","bookSourceName":"A"},{"bookSourceUrl":"http://b.com","bookSourceName":"B"}]`
if got := parseSourcePayload(arr); len(got) != 2 {
if got := ParseSourcePayload(arr); len(got) != 2 {
t.Fatalf("array payload = %d", len(got))
}
// 单对象
single := `{"bookSourceUrl":"http://a.com","bookSourceName":"A"}`
if got := parseSourcePayload(single); len(got) != 1 {
if got := ParseSourcePayload(single); len(got) != 1 {
t.Fatalf("single payload = %d", len(got))
}
// Base64
b64 := base64StdEncode(single)
if got := parseSourcePayload(b64); len(got) != 1 {
if got := ParseSourcePayload(b64); len(got) != 1 {
t.Fatalf("base64 payload = %d", len(got))
}
}
+27
View File
@@ -2,6 +2,7 @@ package rule
import (
"strings"
"time"
"github.com/dlclark/regexp2"
)
@@ -105,6 +106,32 @@ func regexReplaceAll(pattern, result, replacement string) string {
return out
}
// ApplyUserReplace 应用一条用户替换净化规则(对应 legado ReplaceRule)。
// isRegex=false 按字面替换;isRegex=true 用 Java 正则语义并带匹配超时
// (防灾难性回溯挂死服务),编译失败回退字面替换。
func ApplyUserReplace(content, pattern, replacement string, isRegex bool, timeoutMS int64) string {
if pattern == "" {
return content
}
if !isRegex {
return strings.ReplaceAll(content, pattern, replacement)
}
re, err := regexp2.Compile(pattern, regexp2.None)
if err != nil {
return strings.ReplaceAll(content, pattern, replacement)
}
if timeoutMS > 0 {
re.MatchTimeout = time.Duration(timeoutMS) * time.Millisecond
} else {
re.MatchTimeout = 3 * time.Second
}
out, err := re.Replace(content, replacement, 0, -1)
if err != nil {
return content
}
return out
}
// ApplyReplaceRegexString 应用书源 replaceRegex 字符串("##pattern##replace[##x]" 格式),
// 对应 ContentRule.replaceRegex 的处理。
func ApplyReplaceRegexString(content, replaceRegex string) string {
+25 -1
View File
@@ -152,8 +152,32 @@ export const readerAPI = {
listChapters: (id: string) =>
api.get<{ chapters: ReaderChapter[] }>(`/reader/books/${id}/chapters`).then((r) => r.data.chapters),
saveChapters: (id: string, chapters: ReaderChapter[]) => api.post(`/reader/books/${id}/chapters`, { chapters }),
// 书架维度正文(服务端已应用书源 replaceRegex 与用户替换净化规则)
bookContent: (id: string, chapter: number) =>
api
.get<ReaderChapterContent>(`/reader/books/${id}/content`, { params: { chapter }, timeout: LONG_REQUEST_TIMEOUT })
.then((r) => r.data),
// ── 替换规则 ──
// ── 替换净化规则 ──
listReplaceRules: () =>
api.get<{ rules: ReaderReplaceRule[] }>('/reader/replace-rules').then((r) => r.data.rules),
createReplaceRule: (body: ReplaceRuleInput) =>
api.post<ReaderReplaceRule>('/reader/replace-rules', body).then((r) => r.data),
updateReplaceRule: (id: string, body: ReplaceRuleInput) => api.patch(`/reader/replace-rules/${id}`, body),
deleteReplaceRule: (id: string) => api.delete(`/reader/replace-rules/${id}`),
}
export interface ReplaceRuleInput {
name: string
group: string
pattern: string
replacement: string
scope: string
scope_title: boolean
scope_content: boolean
exclude_scope: string
is_enabled: boolean
is_regex: boolean
timeout_millisecond: number
order: number
}
+215
View File
@@ -0,0 +1,215 @@
import { useEffect, useState } from 'react'
import { useNavigate } from 'react-router-dom'
import toast from 'react-hot-toast'
import { ArrowLeft, Loader2, Plus, Trash2 } from 'lucide-react'
import { readerAPI, type ReaderReplaceRule, type ReplaceRuleInput } from '../../api/reader'
// 替换净化规则页(仿 legado ReplaceRuleActivity:列表 + 启停 + 编辑)。
const emptyInput: ReplaceRuleInput = {
name: '',
group: '',
pattern: '',
replacement: '',
scope: '',
scope_title: false,
scope_content: true,
exclude_scope: '',
is_enabled: true,
is_regex: true,
timeout_millisecond: 3000,
order: 0,
}
export default function ReaderReplacePage() {
const navigate = useNavigate()
const [rules, setRules] = useState<ReaderReplaceRule[] | null>(null)
const [showForm, setShowForm] = useState(false)
const [form, setForm] = useState<ReplaceRuleInput>(emptyInput)
const [saving, setSaving] = useState(false)
const load = () => {
readerAPI
.listReplaceRules()
.then(setRules)
.catch(() => toast.error('加载替换规则失败'))
}
// eslint-disable-next-line react-hooks/exhaustive-deps
useEffect(load, [])
const save = async () => {
if (!form.pattern.trim()) {
toast.error('替换规则不能为空')
return
}
setSaving(true)
try {
await readerAPI.createReplaceRule(form)
toast.success('已添加')
setForm(emptyInput)
setShowForm(false)
load()
} catch (e) {
toast.error((e as { response?: { data?: { error?: string } } })?.response?.data?.error ?? '保存失败')
} finally {
setSaving(false)
}
}
const toggle = async (r: ReaderReplaceRule) => {
try {
await readerAPI.updateReplaceRule(r.id, {
name: r.name,
group: r.group,
pattern: r.pattern,
replacement: r.replacement,
scope: r.scope,
scope_title: r.scope_title,
scope_content: r.scope_content,
exclude_scope: r.exclude_scope,
is_enabled: !r.is_enabled,
is_regex: r.is_regex,
timeout_millisecond: 3000,
order: r.order,
})
setRules((prev) => prev?.map((x) => (x.id === r.id ? { ...x, is_enabled: !x.is_enabled } : x)) ?? null)
} catch {
toast.error('更新失败')
}
}
const remove = async (r: ReaderReplaceRule) => {
if (!window.confirm(`删除规则「${r.name || r.pattern}」?`)) return
try {
await readerAPI.deleteReplaceRule(r.id)
setRules((prev) => prev?.filter((x) => x.id !== r.id) ?? null)
} catch {
toast.error('删除失败')
}
}
return (
<div className="mx-auto min-h-[100dvh] w-full max-w-4xl px-4 pb-16 pt-4 sm:px-6">
<div className="flex items-center gap-3">
<button
type="button"
onClick={() => navigate(-1)}
className="rounded-xl p-2 text-[var(--app-muted)] hover:bg-[var(--app-hover)] hover:text-[var(--app-text)]"
>
<ArrowLeft size={18} />
</button>
<h1 className="flex-1 font-display text-lg text-ink-600">替换净化</h1>
<button type="button" onClick={() => setShowForm((v) => !v)} className="btn-primary text-xs">
<Plus size={13} className="mr-1 inline" /> 添加规则
</button>
</div>
{showForm && (
<div className="mt-4 space-y-3 rounded-2xl border border-[var(--app-border)] bg-[var(--app-panel)] p-4">
<div className="grid gap-3 sm:grid-cols-2">
<input
value={form.name}
onChange={(e) => setForm({ ...form, name: e.target.value })}
placeholder="规则名(可选)"
className="rounded-xl border border-[var(--app-border)] bg-[var(--app-panel-soft)] px-3 py-2 text-xs text-[var(--app-text)] outline-none"
/>
<input
value={form.group}
onChange={(e) => setForm({ ...form, group: e.target.value })}
placeholder="分组(可选)"
className="rounded-xl border border-[var(--app-border)] bg-[var(--app-panel-soft)] px-3 py-2 text-xs text-[var(--app-text)] outline-none"
/>
</div>
<textarea
value={form.pattern}
onChange={(e) => setForm({ ...form, pattern: e.target.value })}
rows={2}
placeholder="替换规则(正则或原文)"
className="w-full rounded-xl border border-[var(--app-border)] bg-[var(--app-panel-soft)] p-3 text-xs text-[var(--app-text)] outline-none"
/>
<textarea
value={form.replacement}
onChange={(e) => setForm({ ...form, replacement: e.target.value })}
rows={2}
placeholder="替换为(留空即删除匹配内容)"
className="w-full rounded-xl border border-[var(--app-border)] bg-[var(--app-panel-soft)] p-3 text-xs text-[var(--app-text)] outline-none"
/>
<div className="flex flex-wrap items-center gap-4 text-xs text-[var(--app-muted)]">
<label className="flex items-center gap-1.5">
<input type="checkbox" checked={form.is_regex} onChange={(e) => setForm({ ...form, is_regex: e.target.checked })} />
正则
</label>
<label className="flex items-center gap-1.5">
<input type="checkbox" checked={form.scope_content} onChange={(e) => setForm({ ...form, scope_content: e.target.checked })} />
作用于正文
</label>
<input
value={form.scope}
onChange={(e) => setForm({ ...form, scope: e.target.value })}
placeholder="作用范围(书名包含,可选)"
className="w-44 rounded-xl border border-[var(--app-border)] bg-[var(--app-panel-soft)] px-3 py-1.5 text-xs text-[var(--app-text)] outline-none"
/>
<input
value={form.exclude_scope}
onChange={(e) => setForm({ ...form, exclude_scope: e.target.value })}
placeholder="排除范围(可选)"
className="w-44 rounded-xl border border-[var(--app-border)] bg-[var(--app-panel-soft)] px-3 py-1.5 text-xs text-[var(--app-text)] outline-none"
/>
<button type="button" onClick={save} disabled={saving} className="btn-primary ml-auto text-xs disabled:opacity-50">
{saving ? <Loader2 size={13} className="inline animate-spin" /> : '保存'}
</button>
</div>
</div>
)}
<div className="mt-6 space-y-2">
{rules === null && (
<div className="flex items-center justify-center py-24 text-[var(--app-muted)]">
<Loader2 className="animate-spin" size={22} />
</div>
)}
{rules !== null && rules.length === 0 && (
<div className="rounded-2xl border border-[var(--app-border)] bg-[var(--app-panel-soft)] p-12 text-center text-xs text-[var(--app-muted)]">
还没有替换规则。规则按顺序作用于所有书籍正文,可用来去除广告、修正错字。
</div>
)}
{rules?.map((r) => (
<div key={r.id} className="flex items-center gap-3 rounded-2xl border border-[var(--app-border)] bg-[var(--app-panel)] px-4 py-3">
<div className="min-w-0 flex-1">
<p className="truncate text-sm font-bold text-[var(--app-text)]">
{r.name || '(未命名规则)'}
{r.is_regex && <span className="ml-2 rounded-md bg-brand-500/10 px-1.5 py-0.5 text-2xs font-bold text-brand-600">正则</span>}
{r.group && <span className="ml-2 text-2xs font-normal text-[var(--app-muted)]">{r.group}</span>}
</p>
<p className="mt-0.5 truncate font-mono text-2xs text-[var(--app-muted)]">
{r.pattern} → {r.replacement || '(删除)'}
</p>
</div>
<button
type="button"
title="删除"
onClick={() => remove(r)}
className="rounded-xl p-2 text-[var(--app-muted)] hover:bg-[var(--app-hover)] hover:text-red-500"
>
<Trash2 size={16} />
</button>
<button
type="button"
role="switch"
aria-checked={r.is_enabled}
onClick={() => toggle(r)}
className={`relative h-5 w-9 shrink-0 rounded-full transition ${r.is_enabled ? 'bg-brand-500' : 'bg-[var(--app-hover)]'}`}
>
<span
className={`absolute top-0.5 h-4 w-4 rounded-full bg-white shadow transition-all ${
r.is_enabled ? 'left-[18px]' : 'left-0.5'
}`}
/>
</button>
</div>
))}
</div>
</div>
)
}
+2
View File
@@ -2,6 +2,7 @@ import { Navigate, Route, Routes } from 'react-router-dom'
import ReaderBookPage from './ReaderBookPage'
import ReaderHomePage from './ReaderHomePage'
import ReaderReplacePage from './ReaderReplacePage'
import ReaderSearchPage from './ReaderSearchPage'
import ReaderSourcesPage from './ReaderSourcesPage'
import ReaderViewPage from './ReaderViewPage'
@@ -13,6 +14,7 @@ export default function ReaderRoutes() {
<Route index element={<ReaderHomePage />} />
<Route path="search" element={<ReaderSearchPage />} />
<Route path="sources" element={<ReaderSourcesPage />} />
<Route path="replace" element={<ReaderReplacePage />} />
<Route path="book" element={<ReaderBookPage />} />
<Route path="view/:bookId" element={<ReaderViewPage />} />
<Route path="*" element={<Navigate to="/reader" replace />} />
+4 -1
View File
@@ -1,5 +1,5 @@
import { useEffect, useState } from 'react'
import { useNavigate } from 'react-router-dom'
import { Link, useNavigate } from 'react-router-dom'
import toast from 'react-hot-toast'
import { ArrowLeft, Bug, Download, Loader2, Play, Trash2 } from 'lucide-react'
@@ -102,6 +102,9 @@ export default function ReaderSourcesPage() {
<ArrowLeft size={18} />
</button>
<h1 className="flex-1 font-display text-lg text-ink-600">书源管理</h1>
<Link to="/reader/replace" className="btn-outline mr-2 text-xs">
替换净化
</Link>
<button type="button" onClick={() => setShowImport((v) => !v)} className="btn-primary text-xs">
<Download size={13} className="mr-1 inline" /> 导入书源
</button>
+8 -13
View File
@@ -112,14 +112,11 @@ export default function ReaderViewPage() {
setContent(null)
setLoadingStage('content')
try {
let ct = contentCache.current.get(ch.url)
const cacheKey = String(chapterIndex)
let ct = contentCache.current.get(cacheKey)
if (!ct) {
ct = await readerAPI.content({
source_url: book.origin,
book_url: book.book_url,
chapter_url: ch.url,
})
contentCache.current.set(ch.url, ct)
ct = await readerAPI.bookContent(book.id, chapterIndex)
contentCache.current.set(cacheKey, ct)
}
if (cancelled) return
setContentType(ct.type)
@@ -131,11 +128,10 @@ export default function ReaderViewPage() {
.saveProgress(book.id, { chapter_index: chapterIndex, pos: pendingPosRef.current, chapter_title: ch.title })
.catch(() => undefined)
// 预取下一章
const next = chapters[chapterIndex + 1]
if (next && !contentCache.current.has(next.url)) {
if (!contentCache.current.has(String(chapterIndex + 1))) {
readerAPI
.content({ source_url: book.origin, book_url: book.book_url, chapter_url: next.url })
.then((c) => contentCache.current.set(next.url, c))
.bookContent(book.id, chapterIndex + 1)
.then((c) => contentCache.current.set(String(chapterIndex + 1), c))
.catch(() => undefined)
}
} catch (e) {
@@ -359,8 +355,7 @@ export default function ReaderViewPage() {
onClick={() => {
setError('')
if (chapterIndex !== null) {
const ch = chapters[chapterIndex]
if (ch) contentCache.current.delete(ch.url)
contentCache.current.delete(String(chapterIndex))
setChapterIndex(chapterIndex)
}
}}