diff --git a/internal/service/reader/comment_test.go b/internal/service/reader/comment_test.go
index 62ba7cd..bed9534 100644
--- a/internal/service/reader/comment_test.go
+++ b/internal/service/reader/comment_test.go
@@ -217,9 +217,13 @@ func TestGetContentForBookExtractsComments(t *testing.T) {
if strings.Contains(out.Content, "第一段
第二段
" { + // 块级标签折成换行后,段落文字一个字都不少(多余空行已压掉)。 + if out.Content != "第一段\n第二段" { t.Fatalf("正文被破坏:%q", out.Content) } + if strings.Contains(out.Content, "") || strings.Contains(out.Content, "
") { + t.Fatalf("块级标签没有折行:%q", out.Content) + } if len(out.Comments) != 1 { t.Fatalf("段评数 = %d,期望 1", len(out.Comments)) } @@ -228,3 +232,111 @@ func TestGetContentForBookExtractsComments(t *testing.T) { t.Fatalf("段评内容不符:%+v", c) } } + +// setupTextChapterBook 搭一个「正文规则直接返回指定字符串」的文本源 + 一本书, +// 返回服务实例与书籍 ID,供正文链路回归测试复用。 +func setupTextChapterBook(t *testing.T, content string) (*ReaderService, string) { + t.Helper() + pageJSON := fmt.Sprintf(`{"content":%q}`, content) + srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + w.Header().Set("Content-Type", "application/json; charset=utf-8") + _, _ = w.Write([]byte(pageJSON)) + })) + t.Cleanup(srv.Close) + + svc, _ := newLoginTestService(t) + ctx := t.Context() + srcJSON := fmt.Sprintf(`{ + "bookSourceUrl": %q, + "bookSourceName": "正文链路测试源", + "bookSourceType": 0, + "ruleContent": { "content": "$.content" } +}`, srv.URL) + importTestSource(t, svc, srcJSON, srv.URL) + + book := &model.ReaderBook{ + UserID: "u1", + Origin: srv.URL, + OriginName: "正文链路测试源", + BookURL: srv.URL + "/book/1", + Name: "测试书", + Type: 0, + } + if err := svc.repo.CreateBook(ctx, book); err != nil { + t.Fatalf("创建书籍失败: %v", err) + } + if err := svc.SaveChapters(ctx, book.ID, []ChapterInput{ + {Index: 0, Title: "第一章", URL: srv.URL + "/book/1/c1.html"}, + }); err != nil { + t.Fatalf("写入章节失败: %v", err) + } + return svc, book.ID +} + +// TestGetContentForBookKeepsTextWithStandaloneImage 回归问题 A:带正文的章节 +// 混入一行独立普通插图时,仍然必须是文本类型、正文一个字都不能丢 +// (旧实现把
正文
当孤立标签跳过,整章被误判成图片章清空)。 +func TestGetContentForBookKeepsTextWithStandaloneImage(t *testing.T) { + content := "第一段正文
\n" + + "第二段正文
\n" + + `
`
+ svc, bookID := setupTextChapterBook(t, content)
+
+ out, err := svc.GetContentForBook(t.Context(), "u1", bookID, 0)
+ if err != nil {
+ t.Fatalf("取正文失败: %v", err)
+ }
+ if out.Type != "text" {
+ t.Fatalf("类型 = %q,期望 text(正文不能被当成图片章)", out.Type)
+ }
+ if !strings.Contains(out.Content, "第一段正文") || !strings.Contains(out.Content, "第二段正文") {
+ t.Fatalf("正文文字丢失:%q", out.Content)
+ }
+ if !strings.HasPrefix(out.Content, "第一段正文\n第二段正文\n"+imgMarkerPrefix) {
+ t.Fatalf("正文结构不符:%q", out.Content)
+ }
+ if len(out.Images) != 0 {
+ t.Fatalf("文本类型的 Images 应为空,得到 %v", out.Images)
+ }
+}
+
+// TestGetContentForBookStripsBlockTagsKeepsText 回归问题 B:段评开启时正文是
+// 正文 形态,下发/显示的文本里不能残留块级标签,且文字一字不少。
+func TestGetContentForBookStripsBlockTagsKeepsText(t *testing.T) {
+ src := svgCommentSrc(t, 3, "showCmt('https://v1.example.com/get_review?book_id=1','番茄','段评')", "text")
+ content := "
第一段正文
\n" + + "第二段正文
第三段
折行后的文字
`
+ svc, bookID := setupTextChapterBook(t, content)
+
+ out, err := svc.GetContentForBook(t.Context(), "u1", bookID, 0)
+ if err != nil {
+ t.Fatalf("取正文失败: %v", err)
+ }
+ if out.Type != "text" {
+ t.Fatalf("类型 = %q,期望 text", out.Type)
+ }
+ t.Logf("最终下发/显示的正文 = %q", out.Content)
+ t.Logf("段评锚点 = %+v", out.Comments)
+ for _, tag := range []string{"", "
", "/
/、
、正文
、
。 +var ( + contentBreakRe = regexp.MustCompile(`(?i)<\s*br\s*/?\s*>`) + contentBlockOpenRe = regexp.MustCompile(`(?i)<\s*(?:p|div|h[1-6]|li|tr|blockquote|section|article)\b[^>]*>`) + contentBlockCloseRe = regexp.MustCompile(`(?i)\s*(?:p|div|h[1-6]|li|tr|blockquote|section|article)\s*>`) + contentBlankLineRe = regexp.MustCompile(`\n{2,}`) +) + +// normalizeContentBlocks 把正文里的块级标签(、)会折出多余空行,压成一行;首尾空行一并去掉,
+ // 免得段评行号里混进无意义的空行。
+ content = contentBlankLineRe.ReplaceAllString(content, "\n")
+ return strings.Trim(content, "\n")
+}
+
// rewriteContentImageMarkers 把网络文本正文里「整行就是 」的内容改写成
// [img]<地址> 标记行,前端据此渲染成图片。
//
@@ -1762,7 +1795,8 @@ func rewriteContentImageMarkers(content, baseURL string, proxy func(string) stri
// 一屏只看得到一张,桌面端也没法两页并排——正是「漫画没法双页铺开」的根因。
// 所以这里识别出来之后由调用方把类型改成 image,交给漫画阅读器。
//
-// 只认「非空行全是标记」:混了正文的章节一律保持 text,绝不能把文字吃掉。
+// 只认「非空行全是标记」且至少有一张图:混了正文的章节一律保持 text,绝不能把文字吃掉;
+// 没有一张图标记的空/纯标签章节也不是图片章。
// 返回的地址已经是签名代理地址(调用点在此之前刚做过改写)。
func imageMarkersOnly(content string) ([]string, bool) {
var images []string
@@ -1785,30 +1819,47 @@ func imageMarkersOnly(content string) ([]string, bool) {
return images, len(images) > 0
}
-// isHTMLTagOnly 判断整行是不是一个孤立的 HTML 标签(
正文
这类「标签里裹着正文」的行第一个 '>' 后面还有文字, +// 必须当正文放行——旧实现看到标签名后的第一个非字母字符('>')就返回 true, +// 于是整章正文被 imageMarkersOnly 当成孤立标签跳过、误判成图片章清空。 +// +// 尖括号里的内容必须像标签(字母开头)才算数,避免把「3 < 5」这类普通文字行当成标签。 func isHTMLTagOnly(line string) bool { - if len(line) < 3 || line[0] != '<' || line[len(line)-1] != '>' { + if !strings.ContainsRune(line, '<') { return false } - inner := strings.TrimSpace(strings.TrimPrefix(line[1:len(line)-1], "/")) + for i := 0; i < len(line); { + if line[i] != '<' { + if !isHTMLSpace(line[i]) { + return false // 尖括号外还有可读文字 + } + i++ + continue + } + gt := strings.IndexByte(line[i:], '>') + if gt < 0 { + return false // 没闭合的 '<',当正文处理 + } + if !looksLikeTag(line[i+1 : i+gt]) { + return false + } + i += gt + 1 + } + return true +} + +// looksLikeTag 判断尖括号里的内容是不是标签名(正文
是「标签里裹着正文」,绝不能当成孤立标签跳过, + // 否则整章正文会被误判成图片章清空。 + name: "标签里裹着正文的行", + content: imgMarkerPrefix + "/a.jpg\n第一章正文
\n第二章正文
\n", + wantOK: false, + }, } for _, c := range cases { got, ok := imageMarkersOnly(c.content) @@ -127,6 +134,46 @@ func TestImageMarkersOnly(t *testing.T) { } } +// TestIsHTMLTagOnly 「没有可读文字的标签行」才算空行:标签里裹着正文的行必须放行, +// 否则 imageMarkersOnly 会把整章正文当图片丢掉。 +func TestIsHTMLTagOnly(t *testing.T) { + blank := []string{""} + for _, s := range blank { + if !isHTMLTagOnly(s) { + t.Errorf("%q 应视为没有可读文字的标签行", s) + } + } + text := []string{ + "
正文
", "正文
", "第一段
第二段
", + "正文 粗", "3 < 5", "<3>", "a", "", " ", + } + for _, s := range text { + if isHTMLTagOnly(s) { + t.Errorf("%q 含可读文字/不是标签,不能当空行", s) + } + } +} + +// TestNormalizeContentBlocks 块级标签折成换行、行内标签与文字原样保留。 +func TestNormalizeContentBlocks(t *testing.T) { + cases := []struct{ in, want string }{ + {"第一段
\n第二段
", "第一段\n第二段"}, + {"第一段
折行文字
大写带属性
", "大写带属性"}, + {"正文
`},
+ // 纯文本原样返回
+ {"第一段。\n第二段。", "第一段。\n第二段。"},
+ {"", ""},
+ }
+ for _, c := range cases {
+ if got := normalizeContentBlocks(c.in); got != c.want {
+ t.Errorf("normalizeContentBlocks(%q) = %q,期望 %q", c.in, got, c.want)
+ }
+ }
+}
+
// TestSearchCheckKeyWord 校验关键字的取值规则(对应 legado getCheckKeyword):
// 含 http/::/++/-- 的值是地址或扩展标记,不当作关键字。
func TestSearchCheckKeyWord(t *testing.T) {