diff --git a/internal/service/reader/comment_test.go b/internal/service/reader/comment_test.go index 62ba7cd..bed9534 100644 --- a/internal/service/reader/comment_test.go +++ b/internal/service/reader/comment_test.go @@ -217,9 +217,13 @@ func TestGetContentForBookExtractsComments(t *testing.T) { if strings.Contains(out.Content, "") || strings.Contains(out.Content, "

") { + t.Fatalf("块级标签没有折行:%q", out.Content) + } if len(out.Comments) != 1 { t.Fatalf("段评数 = %d,期望 1", len(out.Comments)) } @@ -228,3 +232,111 @@ func TestGetContentForBookExtractsComments(t *testing.T) { t.Fatalf("段评内容不符:%+v", c) } } + +// setupTextChapterBook 搭一个「正文规则直接返回指定字符串」的文本源 + 一本书, +// 返回服务实例与书籍 ID,供正文链路回归测试复用。 +func setupTextChapterBook(t *testing.T, content string) (*ReaderService, string) { + t.Helper() + pageJSON := fmt.Sprintf(`{"content":%q}`, content) + srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + w.Header().Set("Content-Type", "application/json; charset=utf-8") + _, _ = w.Write([]byte(pageJSON)) + })) + t.Cleanup(srv.Close) + + svc, _ := newLoginTestService(t) + ctx := t.Context() + srcJSON := fmt.Sprintf(`{ + "bookSourceUrl": %q, + "bookSourceName": "正文链路测试源", + "bookSourceType": 0, + "ruleContent": { "content": "$.content" } +}`, srv.URL) + importTestSource(t, svc, srcJSON, srv.URL) + + book := &model.ReaderBook{ + UserID: "u1", + Origin: srv.URL, + OriginName: "正文链路测试源", + BookURL: srv.URL + "/book/1", + Name: "测试书", + Type: 0, + } + if err := svc.repo.CreateBook(ctx, book); err != nil { + t.Fatalf("创建书籍失败: %v", err) + } + if err := svc.SaveChapters(ctx, book.ID, []ChapterInput{ + {Index: 0, Title: "第一章", URL: srv.URL + "/book/1/c1.html"}, + }); err != nil { + t.Fatalf("写入章节失败: %v", err) + } + return svc, book.ID +} + +// TestGetContentForBookKeepsTextWithStandaloneImage 回归问题 A:带

正文的章节 +// 混入一行独立普通插图时,仍然必须是文本类型、正文一个字都不能丢 +// (旧实现把

正文

当孤立标签跳过,整章被误判成图片章清空)。 +func TestGetContentForBookKeepsTextWithStandaloneImage(t *testing.T) { + content := "

第一段正文

\n" + + "

第二段正文

\n" + + `` + svc, bookID := setupTextChapterBook(t, content) + + out, err := svc.GetContentForBook(t.Context(), "u1", bookID, 0) + if err != nil { + t.Fatalf("取正文失败: %v", err) + } + if out.Type != "text" { + t.Fatalf("类型 = %q,期望 text(正文不能被当成图片章)", out.Type) + } + if !strings.Contains(out.Content, "第一段正文") || !strings.Contains(out.Content, "第二段正文") { + t.Fatalf("正文文字丢失:%q", out.Content) + } + if !strings.HasPrefix(out.Content, "第一段正文\n第二段正文\n"+imgMarkerPrefix) { + t.Fatalf("正文结构不符:%q", out.Content) + } + if len(out.Images) != 0 { + t.Fatalf("文本类型的 Images 应为空,得到 %v", out.Images) + } +} + +// TestGetContentForBookStripsBlockTagsKeepsText 回归问题 B:段评开启时正文是 +//

正文 形态,下发/显示的文本里不能残留块级标签,且文字一字不少。 +func TestGetContentForBookStripsBlockTagsKeepsText(t *testing.T) { + src := svgCommentSrc(t, 3, "showCmt('https://v1.example.com/get_review?book_id=1','番茄','段评')", "text") + content := "

第一段正文

\n" + + "

第二段正文

\n" + + "

第三段
折行后的文字

\n" + + `` + svc, bookID := setupTextChapterBook(t, content) + + out, err := svc.GetContentForBook(t.Context(), "u1", bookID, 0) + if err != nil { + t.Fatalf("取正文失败: %v", err) + } + if out.Type != "text" { + t.Fatalf("类型 = %q,期望 text", out.Type) + } + t.Logf("最终下发/显示的正文 = %q", out.Content) + t.Logf("段评锚点 = %+v", out.Comments) + for _, tag := range []string{"

", "

", "
", " 折出来的那一段)。 + for _, want := range []string{"第一段正文", "第二段正文", "折行后的文字"} { + if !strings.Contains(out.Content, want) { + t.Fatalf("正文文字丢失 %q:%q", want, out.Content) + } + } + if !strings.Contains(out.Content, imgMarkerPrefix) { + t.Fatalf("普通插图应保留成 %s 标记:%q", imgMarkerPrefix, out.Content) + } + if len(out.Comments) != 1 { + t.Fatalf("段评数 = %d,期望 1", len(out.Comments)) + } + if out.Comments[0].URL != "https://v1.example.com/get_review?book_id=1" { + t.Fatalf("段评地址不符:%+v", out.Comments[0]) + } +} diff --git a/internal/service/reader/reader.go b/internal/service/reader/reader.go index ac91009..f7cc477 100644 --- a/internal/service/reader/reader.go +++ b/internal/service/reader/reader.go @@ -13,6 +13,7 @@ import ( "io" "net/http" "net/url" + "regexp" "sort" "strings" "sync" @@ -1707,6 +1708,9 @@ func (s *ReaderService) getContentFrom(ctx context.Context, src *model.ReaderBoo out.ImageStyle = SPtr(cr.ImageStyle) default: out.Type = "text" + // 块级标签(

/

/
)折成换行:书源段评开启时正文就是这种 HTML 形态, + // 前端纯文本渲染,不折的话读者会看到字面标签。必须先折行再摘段评(行号会变)。 + content = normalizeContentBlocks(content) // 正文里的段评标记( 或「图片地址 + click 配置」的内嵌评论图) // 归一成结构化锚点,正文文字原样保留(见 comment.go 的说明)。 out.Content, out.Comments = extractContentComments(content) @@ -1714,6 +1718,35 @@ func (s *ReaderService) getContentFrom(ctx context.Context, src *model.ReaderBoo return out, nil } +// 正文里的块级标签边界:

、

、
这类只表达段落、没有文字的标签。 +// 书源(如光遇聚合的 paraForAndroid)在段评开启时把正文拼成

正文

, +// 前端是纯文本渲染({text}),不折行的话读者看到的就是字面的

、

。 +var ( + contentBreakRe = regexp.MustCompile(`(?i)<\s*br\s*/?\s*>`) + contentBlockOpenRe = regexp.MustCompile(`(?i)<\s*(?:p|div|h[1-6]|li|tr|blockquote|section|article)\b[^>]*>`) + contentBlockCloseRe = regexp.MustCompile(`(?i)`) + contentBlankLineRe = regexp.MustCompile(`\n{2,}`) +) + +// normalizeContentBlocks 把正文里的块级标签(

、
等)按语义折成换行。 +// +// 只处理段落边界的标签,绝不删除行内标签(普通插图 原样保留), +// 也不做任意 HTML 渲染——前端仍然只渲染纯文本 + [img] 标记。 +// 必须在摘段评之前调用:折行会改变行号,段评锚点要落在折行后的正文上, +// 否则前端按行号挂气泡会错位。 +func normalizeContentBlocks(content string) string { + if !strings.ContainsRune(content, '<') { + return content + } + content = contentBreakRe.ReplaceAllString(content, "\n") + content = contentBlockOpenRe.ReplaceAllString(content, "\n") + content = contentBlockCloseRe.ReplaceAllString(content, "\n") + // 相邻块边界(

与下一行

)会折出多余空行,压成一行;首尾空行一并去掉, + // 免得段评行号里混进无意义的空行。 + content = contentBlankLineRe.ReplaceAllString(content, "\n") + return strings.Trim(content, "\n") +} + // rewriteContentImageMarkers 把网络文本正文里「整行就是 」的内容改写成 // [img]<地址> 标记行,前端据此渲染成图片。 // @@ -1762,7 +1795,8 @@ func rewriteContentImageMarkers(content, baseURL string, proxy func(string) stri // 一屏只看得到一张,桌面端也没法两页并排——正是「漫画没法双页铺开」的根因。 // 所以这里识别出来之后由调用方把类型改成 image,交给漫画阅读器。 // -// 只认「非空行全是标记」:混了正文的章节一律保持 text,绝不能把文字吃掉。 +// 只认「非空行全是标记」且至少有一张图:混了正文的章节一律保持 text,绝不能把文字吃掉; +// 没有一张图标记的空/纯标签章节也不是图片章。 // 返回的地址已经是签名代理地址(调用点在此之前刚做过改写)。 func imageMarkersOnly(content string) ([]string, bool) { var images []string @@ -1785,30 +1819,47 @@ func imageMarkersOnly(content string) ([]string, bool) { return images, len(images) > 0 } -// isHTMLTagOnly 判断整行是不是一个孤立的 HTML 标签(

、
、
之类)。 -// 即「<」开头、「>」结尾,且尖括号内是标签名的形状(字母开头,后面可带属性)。 -// 标签名两侧的空白(源码里常见的 < div > 这类手写残渣)一并忽略。 +// isHTMLTagOnly 判断整行是不是「没有可读文字的标签/空白」(
、
、< br > 之类)。 +// +// 逐个跳过尖括号片段,只要尖括号外还剩非空白字符,就不是空行。 +// 关键:

正文

这类「标签里裹着正文」的行第一个 '>' 后面还有文字, +// 必须当正文放行——旧实现看到标签名后的第一个非字母字符('>')就返回 true, +// 于是整章正文被 imageMarkersOnly 当成孤立标签跳过、误判成图片章清空。 +// +// 尖括号里的内容必须像标签(字母开头)才算数,避免把「3 < 5」这类普通文字行当成标签。 func isHTMLTagOnly(line string) bool { - if len(line) < 3 || line[0] != '<' || line[len(line)-1] != '>' { + if !strings.ContainsRune(line, '<') { return false } - inner := strings.TrimSpace(strings.TrimPrefix(line[1:len(line)-1], "/")) + for i := 0; i < len(line); { + if line[i] != '<' { + if !isHTMLSpace(line[i]) { + return false // 尖括号外还有可读文字 + } + i++ + continue + } + gt := strings.IndexByte(line[i:], '>') + if gt < 0 { + return false // 没闭合的 '<',当正文处理 + } + if !looksLikeTag(line[i+1 : i+gt]) { + return false + } + i += gt + 1 + } + return true +} + +// looksLikeTag 判断尖括号里的内容是不是标签名(
、
、< br >), +// 而不是「<3」这类普通文字。允许标签名前后的空白与自闭合斜杠。 +func looksLikeTag(inner string) bool { + inner = strings.TrimSpace(strings.TrimPrefix(strings.TrimSpace(inner), "/")) if inner == "" { return false } - for i, r := range inner { - switch { - case (r >= 'a' && r <= 'z') || (r >= 'A' && r <= 'Z'): - continue - case i == 0: - // 首字符不是字母:<3、 之类,一律不当标签 - return false - default: - // 标签名之后(属性、空白、自闭合斜杠)都算标签 - return true - } - } - return true + c := inner[0] + return (c >= 'a' && c <= 'z') || (c >= 'A' && c <= 'Z') } // imageRefsInLine 取出一行正文里的图片地址。 diff --git a/internal/service/reader/reader_helpers_test.go b/internal/service/reader/reader_helpers_test.go index 2923943..65d66a6 100644 --- a/internal/service/reader/reader_helpers_test.go +++ b/internal/service/reader/reader_helpers_test.go @@ -105,6 +105,13 @@ func TestImageMarkersOnly(t *testing.T) { content: imgMarkerPrefix + "/a.jpg\n3 < 5\n", wantOK: false, }, + { + // 回归:

正文

是「标签里裹着正文」,绝不能当成孤立标签跳过, + // 否则整章正文会被误判成图片章清空。 + name: "标签里裹着正文的行", + content: imgMarkerPrefix + "/a.jpg\n

第一章正文

\n

第二章正文

\n", + wantOK: false, + }, } for _, c := range cases { got, ok := imageMarkersOnly(c.content) @@ -127,6 +134,46 @@ func TestImageMarkersOnly(t *testing.T) { } } +// TestIsHTMLTagOnly 「没有可读文字的标签行」才算空行:标签里裹着正文的行必须放行, +// 否则 imageMarkersOnly 会把整章正文当图片丢掉。 +func TestIsHTMLTagOnly(t *testing.T) { + blank := []string{"
", "
", "< br >", "
", "

", "
", "

", " ", "

"} + for _, s := range blank { + if !isHTMLTagOnly(s) { + t.Errorf("%q 应视为没有可读文字的标签行", s) + } + } + text := []string{ + "

正文

", "

正文

", "
文字
", "

第一段

第二段

", + "正文 粗", "3 < 5", "<3>", "a", "", " ", + } + for _, s := range text { + if isHTMLTagOnly(s) { + t.Errorf("%q 含可读文字/不是标签,不能当空行", s) + } + } +} + +// TestNormalizeContentBlocks 块级标签折成换行、行内标签与文字原样保留。 +func TestNormalizeContentBlocks(t *testing.T) { + cases := []struct{ in, want string }{ + {"

第一段

\n

第二段

", "第一段\n第二段"}, + {"

第一段
折行文字

", "第一段\n折行文字"}, + {"

大写带属性

", "大写带属性"}, + {"
块


", "块"}, + // 行内插图原样保留(不能吃掉普通 ) + {`

正文

`, `正文`}, + // 纯文本原样返回 + {"第一段。\n第二段。", "第一段。\n第二段。"}, + {"", ""}, + } + for _, c := range cases { + if got := normalizeContentBlocks(c.in); got != c.want { + t.Errorf("normalizeContentBlocks(%q) = %q,期望 %q", c.in, got, c.want) + } + } +} + // TestSearchCheckKeyWord 校验关键字的取值规则(对应 legado getCheckKeyword): // 含 http/::/++/-- 的值是地址或扩展标记,不当作关键字。 func TestSearchCheckKeyWord(t *testing.T) {