mirror of
https://github.com/truewhile/MeBox.git
synced 2026-10-04 04:26:38 +08:00
优化添加段评内容
This commit is contained in:
@@ -217,9 +217,13 @@ func TestGetContentForBookExtractsComments(t *testing.T) {
|
||||
if strings.Contains(out.Content, "<img") || strings.Contains(out.Content, "showCmt") {
|
||||
t.Fatalf("段评标记没有清干净:%q", out.Content)
|
||||
}
|
||||
if out.Content != "<p>第一段</p>\n<p>第二段</p>" {
|
||||
// 块级标签折成换行后,段落文字一个字都不少(多余空行已压掉)。
|
||||
if out.Content != "第一段\n第二段" {
|
||||
t.Fatalf("正文被破坏:%q", out.Content)
|
||||
}
|
||||
if strings.Contains(out.Content, "<p>") || strings.Contains(out.Content, "</p>") {
|
||||
t.Fatalf("块级标签没有折行:%q", out.Content)
|
||||
}
|
||||
if len(out.Comments) != 1 {
|
||||
t.Fatalf("段评数 = %d,期望 1", len(out.Comments))
|
||||
}
|
||||
@@ -228,3 +232,111 @@ func TestGetContentForBookExtractsComments(t *testing.T) {
|
||||
t.Fatalf("段评内容不符:%+v", c)
|
||||
}
|
||||
}
|
||||
|
||||
// setupTextChapterBook 搭一个「正文规则直接返回指定字符串」的文本源 + 一本书,
|
||||
// 返回服务实例与书籍 ID,供正文链路回归测试复用。
|
||||
func setupTextChapterBook(t *testing.T, content string) (*ReaderService, string) {
|
||||
t.Helper()
|
||||
pageJSON := fmt.Sprintf(`{"content":%q}`, content)
|
||||
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
w.Header().Set("Content-Type", "application/json; charset=utf-8")
|
||||
_, _ = w.Write([]byte(pageJSON))
|
||||
}))
|
||||
t.Cleanup(srv.Close)
|
||||
|
||||
svc, _ := newLoginTestService(t)
|
||||
ctx := t.Context()
|
||||
srcJSON := fmt.Sprintf(`{
|
||||
"bookSourceUrl": %q,
|
||||
"bookSourceName": "正文链路测试源",
|
||||
"bookSourceType": 0,
|
||||
"ruleContent": { "content": "$.content" }
|
||||
}`, srv.URL)
|
||||
importTestSource(t, svc, srcJSON, srv.URL)
|
||||
|
||||
book := &model.ReaderBook{
|
||||
UserID: "u1",
|
||||
Origin: srv.URL,
|
||||
OriginName: "正文链路测试源",
|
||||
BookURL: srv.URL + "/book/1",
|
||||
Name: "测试书",
|
||||
Type: 0,
|
||||
}
|
||||
if err := svc.repo.CreateBook(ctx, book); err != nil {
|
||||
t.Fatalf("创建书籍失败: %v", err)
|
||||
}
|
||||
if err := svc.SaveChapters(ctx, book.ID, []ChapterInput{
|
||||
{Index: 0, Title: "第一章", URL: srv.URL + "/book/1/c1.html"},
|
||||
}); err != nil {
|
||||
t.Fatalf("写入章节失败: %v", err)
|
||||
}
|
||||
return svc, book.ID
|
||||
}
|
||||
|
||||
// TestGetContentForBookKeepsTextWithStandaloneImage 回归问题 A:带 <p> 正文的章节
|
||||
// 混入一行独立普通插图时,仍然必须是文本类型、正文一个字都不能丢
|
||||
// (旧实现把 <p>正文</p> 当孤立标签跳过,整章被误判成图片章清空)。
|
||||
func TestGetContentForBookKeepsTextWithStandaloneImage(t *testing.T) {
|
||||
content := "<p>第一段正文</p>\n" +
|
||||
"<p>第二段正文</p>\n" +
|
||||
`<img src="https://cdn.example.com/pic.jpg">`
|
||||
svc, bookID := setupTextChapterBook(t, content)
|
||||
|
||||
out, err := svc.GetContentForBook(t.Context(), "u1", bookID, 0)
|
||||
if err != nil {
|
||||
t.Fatalf("取正文失败: %v", err)
|
||||
}
|
||||
if out.Type != "text" {
|
||||
t.Fatalf("类型 = %q,期望 text(正文不能被当成图片章)", out.Type)
|
||||
}
|
||||
if !strings.Contains(out.Content, "第一段正文") || !strings.Contains(out.Content, "第二段正文") {
|
||||
t.Fatalf("正文文字丢失:%q", out.Content)
|
||||
}
|
||||
if !strings.HasPrefix(out.Content, "第一段正文\n第二段正文\n"+imgMarkerPrefix) {
|
||||
t.Fatalf("正文结构不符:%q", out.Content)
|
||||
}
|
||||
if len(out.Images) != 0 {
|
||||
t.Fatalf("文本类型的 Images 应为空,得到 %v", out.Images)
|
||||
}
|
||||
}
|
||||
|
||||
// TestGetContentForBookStripsBlockTagsKeepsText 回归问题 B:段评开启时正文是
|
||||
// <p>正文<img …,{click}> 形态,下发/显示的文本里不能残留块级标签,且文字一字不少。
|
||||
func TestGetContentForBookStripsBlockTagsKeepsText(t *testing.T) {
|
||||
src := svgCommentSrc(t, 3, "showCmt('https://v1.example.com/get_review?book_id=1','番茄','段评')", "text")
|
||||
content := "<p>第一段正文</p>\n" +
|
||||
"<p>第二段正文<img src=\"" + src + "\"></p>\n" +
|
||||
"<p>第三段<br>折行后的文字</p>\n" +
|
||||
`<img src="https://cdn.example.com/pic.jpg">`
|
||||
svc, bookID := setupTextChapterBook(t, content)
|
||||
|
||||
out, err := svc.GetContentForBook(t.Context(), "u1", bookID, 0)
|
||||
if err != nil {
|
||||
t.Fatalf("取正文失败: %v", err)
|
||||
}
|
||||
if out.Type != "text" {
|
||||
t.Fatalf("类型 = %q,期望 text", out.Type)
|
||||
}
|
||||
t.Logf("最终下发/显示的正文 = %q", out.Content)
|
||||
t.Logf("段评锚点 = %+v", out.Comments)
|
||||
for _, tag := range []string{"<p>", "</p>", "<br>", "<img", "showCmt"} {
|
||||
if strings.Contains(out.Content, tag) {
|
||||
t.Fatalf("显示文本里残留了 %q:%q", tag, out.Content)
|
||||
}
|
||||
}
|
||||
// 正文文字一字不少(含 <br> 折出来的那一段)。
|
||||
for _, want := range []string{"第一段正文", "第二段正文", "折行后的文字"} {
|
||||
if !strings.Contains(out.Content, want) {
|
||||
t.Fatalf("正文文字丢失 %q:%q", want, out.Content)
|
||||
}
|
||||
}
|
||||
if !strings.Contains(out.Content, imgMarkerPrefix) {
|
||||
t.Fatalf("普通插图应保留成 %s 标记:%q", imgMarkerPrefix, out.Content)
|
||||
}
|
||||
if len(out.Comments) != 1 {
|
||||
t.Fatalf("段评数 = %d,期望 1", len(out.Comments))
|
||||
}
|
||||
if out.Comments[0].URL != "https://v1.example.com/get_review?book_id=1" {
|
||||
t.Fatalf("段评地址不符:%+v", out.Comments[0])
|
||||
}
|
||||
}
|
||||
|
||||
@@ -13,6 +13,7 @@ import (
|
||||
"io"
|
||||
"net/http"
|
||||
"net/url"
|
||||
"regexp"
|
||||
"sort"
|
||||
"strings"
|
||||
"sync"
|
||||
@@ -1707,6 +1708,9 @@ func (s *ReaderService) getContentFrom(ctx context.Context, src *model.ReaderBoo
|
||||
out.ImageStyle = SPtr(cr.ImageStyle)
|
||||
default:
|
||||
out.Type = "text"
|
||||
// 块级标签(<p>/</p>/<br>)折成换行:书源段评开启时正文就是这种 HTML 形态,
|
||||
// 前端纯文本渲染,不折的话读者会看到字面标签。必须先折行再摘段评(行号会变)。
|
||||
content = normalizeContentBlocks(content)
|
||||
// 正文里的段评标记(<comment …/> 或「图片地址 + click 配置」的内嵌评论图)
|
||||
// 归一成结构化锚点,正文文字原样保留(见 comment.go 的说明)。
|
||||
out.Content, out.Comments = extractContentComments(content)
|
||||
@@ -1714,6 +1718,35 @@ func (s *ReaderService) getContentFrom(ctx context.Context, src *model.ReaderBoo
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// 正文里的块级标签边界:<p>、</p>、<br> 这类只表达段落、没有文字的标签。
|
||||
// 书源(如光遇聚合的 paraForAndroid)在段评开启时把正文拼成 <p>正文<comment/></p>,
|
||||
// 前端是纯文本渲染({text}),不折行的话读者看到的就是字面的 <p>、</p>。
|
||||
var (
|
||||
contentBreakRe = regexp.MustCompile(`(?i)<\s*br\s*/?\s*>`)
|
||||
contentBlockOpenRe = regexp.MustCompile(`(?i)<\s*(?:p|div|h[1-6]|li|tr|blockquote|section|article)\b[^>]*>`)
|
||||
contentBlockCloseRe = regexp.MustCompile(`(?i)</\s*(?:p|div|h[1-6]|li|tr|blockquote|section|article)\s*>`)
|
||||
contentBlankLineRe = regexp.MustCompile(`\n{2,}`)
|
||||
)
|
||||
|
||||
// normalizeContentBlocks 把正文里的块级标签(<p></p>、<br> 等)按语义折成换行。
|
||||
//
|
||||
// 只处理段落边界的标签,绝不删除行内标签(普通插图 <img> 原样保留),
|
||||
// 也不做任意 HTML 渲染——前端仍然只渲染纯文本 + [img] 标记。
|
||||
// 必须在摘段评之前调用:折行会改变行号,段评锚点要落在折行后的正文上,
|
||||
// 否则前端按行号挂气泡会错位。
|
||||
func normalizeContentBlocks(content string) string {
|
||||
if !strings.ContainsRune(content, '<') {
|
||||
return content
|
||||
}
|
||||
content = contentBreakRe.ReplaceAllString(content, "\n")
|
||||
content = contentBlockOpenRe.ReplaceAllString(content, "\n")
|
||||
content = contentBlockCloseRe.ReplaceAllString(content, "\n")
|
||||
// 相邻块边界(</p> 与下一行 <p>)会折出多余空行,压成一行;首尾空行一并去掉,
|
||||
// 免得段评行号里混进无意义的空行。
|
||||
content = contentBlankLineRe.ReplaceAllString(content, "\n")
|
||||
return strings.Trim(content, "\n")
|
||||
}
|
||||
|
||||
// rewriteContentImageMarkers 把网络文本正文里「整行就是 <img src="…">」的内容改写成
|
||||
// [img]<地址> 标记行,前端据此渲染成图片。
|
||||
//
|
||||
@@ -1762,7 +1795,8 @@ func rewriteContentImageMarkers(content, baseURL string, proxy func(string) stri
|
||||
// 一屏只看得到一张,桌面端也没法两页并排——正是「漫画没法双页铺开」的根因。
|
||||
// 所以这里识别出来之后由调用方把类型改成 image,交给漫画阅读器。
|
||||
//
|
||||
// 只认「非空行全是标记」:混了正文的章节一律保持 text,绝不能把文字吃掉。
|
||||
// 只认「非空行全是标记」且至少有一张图:混了正文的章节一律保持 text,绝不能把文字吃掉;
|
||||
// 没有一张图标记的空/纯标签章节也不是图片章。
|
||||
// 返回的地址已经是签名代理地址(调用点在此之前刚做过改写)。
|
||||
func imageMarkersOnly(content string) ([]string, bool) {
|
||||
var images []string
|
||||
@@ -1785,30 +1819,47 @@ func imageMarkersOnly(content string) ([]string, bool) {
|
||||
return images, len(images) > 0
|
||||
}
|
||||
|
||||
// isHTMLTagOnly 判断整行是不是一个孤立的 HTML 标签(<div>、</div>、<br> 之类)。
|
||||
// 即「<」开头、「>」结尾,且尖括号内是标签名的形状(字母开头,后面可带属性)。
|
||||
// 标签名两侧的空白(源码里常见的 < div > 这类手写残渣)一并忽略。
|
||||
// isHTMLTagOnly 判断整行是不是「没有可读文字的标签/空白」(<div>、</div>、< br > 之类)。
|
||||
//
|
||||
// 逐个跳过尖括号片段,只要尖括号外还剩非空白字符,就不是空行。
|
||||
// 关键:<p>正文</p> 这类「标签里裹着正文」的行第一个 '>' 后面还有文字,
|
||||
// 必须当正文放行——旧实现看到标签名后的第一个非字母字符('>')就返回 true,
|
||||
// 于是整章正文被 imageMarkersOnly 当成孤立标签跳过、误判成图片章清空。
|
||||
//
|
||||
// 尖括号里的内容必须像标签(字母开头)才算数,避免把「3 < 5」这类普通文字行当成标签。
|
||||
func isHTMLTagOnly(line string) bool {
|
||||
if len(line) < 3 || line[0] != '<' || line[len(line)-1] != '>' {
|
||||
if !strings.ContainsRune(line, '<') {
|
||||
return false
|
||||
}
|
||||
inner := strings.TrimSpace(strings.TrimPrefix(line[1:len(line)-1], "/"))
|
||||
for i := 0; i < len(line); {
|
||||
if line[i] != '<' {
|
||||
if !isHTMLSpace(line[i]) {
|
||||
return false // 尖括号外还有可读文字
|
||||
}
|
||||
i++
|
||||
continue
|
||||
}
|
||||
gt := strings.IndexByte(line[i:], '>')
|
||||
if gt < 0 {
|
||||
return false // 没闭合的 '<',当正文处理
|
||||
}
|
||||
if !looksLikeTag(line[i+1 : i+gt]) {
|
||||
return false
|
||||
}
|
||||
i += gt + 1
|
||||
}
|
||||
return true
|
||||
}
|
||||
|
||||
// looksLikeTag 判断尖括号里的内容是不是标签名(<div>、</div>、< br >),
|
||||
// 而不是「<3」这类普通文字。允许标签名前后的空白与自闭合斜杠。
|
||||
func looksLikeTag(inner string) bool {
|
||||
inner = strings.TrimSpace(strings.TrimPrefix(strings.TrimSpace(inner), "/"))
|
||||
if inner == "" {
|
||||
return false
|
||||
}
|
||||
for i, r := range inner {
|
||||
switch {
|
||||
case (r >= 'a' && r <= 'z') || (r >= 'A' && r <= 'Z'):
|
||||
continue
|
||||
case i == 0:
|
||||
// 首字符不是字母:<3、<!-- 注释 --> 之类,一律不当标签
|
||||
return false
|
||||
default:
|
||||
// 标签名之后(属性、空白、自闭合斜杠)都算标签
|
||||
return true
|
||||
}
|
||||
}
|
||||
return true
|
||||
c := inner[0]
|
||||
return (c >= 'a' && c <= 'z') || (c >= 'A' && c <= 'Z')
|
||||
}
|
||||
|
||||
// imageRefsInLine 取出一行正文里的图片地址。
|
||||
|
||||
@@ -105,6 +105,13 @@ func TestImageMarkersOnly(t *testing.T) {
|
||||
content: imgMarkerPrefix + "/a.jpg\n3 < 5\n",
|
||||
wantOK: false,
|
||||
},
|
||||
{
|
||||
// 回归:<p>正文</p> 是「标签里裹着正文」,绝不能当成孤立标签跳过,
|
||||
// 否则整章正文会被误判成图片章清空。
|
||||
name: "标签里裹着正文的行",
|
||||
content: imgMarkerPrefix + "/a.jpg\n<p>第一章正文</p>\n<p>第二章正文</p>\n",
|
||||
wantOK: false,
|
||||
},
|
||||
}
|
||||
for _, c := range cases {
|
||||
got, ok := imageMarkersOnly(c.content)
|
||||
@@ -127,6 +134,46 @@ func TestImageMarkersOnly(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// TestIsHTMLTagOnly 「没有可读文字的标签行」才算空行:标签里裹着正文的行必须放行,
|
||||
// 否则 imageMarkersOnly 会把整章正文当图片丢掉。
|
||||
func TestIsHTMLTagOnly(t *testing.T) {
|
||||
blank := []string{"<div>", "</div>", "< br >", "<br/>", "<p></p>", "<hr />", "</p >", "<a> </a>", "<p>"}
|
||||
for _, s := range blank {
|
||||
if !isHTMLTagOnly(s) {
|
||||
t.Errorf("%q 应视为没有可读文字的标签行", s)
|
||||
}
|
||||
}
|
||||
text := []string{
|
||||
"<p>正文</p>", "<p>正文</p></p>", "<div>文字</div>", "<p>第一段</p><p>第二段</p>",
|
||||
"正文 <b>粗</b>", "3 < 5", "<3>", "a", "", " ",
|
||||
}
|
||||
for _, s := range text {
|
||||
if isHTMLTagOnly(s) {
|
||||
t.Errorf("%q 含可读文字/不是标签,不能当空行", s)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestNormalizeContentBlocks 块级标签折成换行、行内标签与文字原样保留。
|
||||
func TestNormalizeContentBlocks(t *testing.T) {
|
||||
cases := []struct{ in, want string }{
|
||||
{"<p>第一段</p>\n<p>第二段</p>", "第一段\n第二段"},
|
||||
{"<p>第一段<br>折行文字</p>", "第一段\n折行文字"},
|
||||
{"<P CLASS='x'>大写带属性</P>", "大写带属性"},
|
||||
{"<div>块</div><br/><br />", "块"},
|
||||
// 行内插图原样保留(不能吃掉普通 <img>)
|
||||
{`<p>正文<img src="https://cdn.example.com/a.jpg"></p>`, `正文<img src="https://cdn.example.com/a.jpg">`},
|
||||
// 纯文本原样返回
|
||||
{"第一段。\n第二段。", "第一段。\n第二段。"},
|
||||
{"", ""},
|
||||
}
|
||||
for _, c := range cases {
|
||||
if got := normalizeContentBlocks(c.in); got != c.want {
|
||||
t.Errorf("normalizeContentBlocks(%q) = %q,期望 %q", c.in, got, c.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestSearchCheckKeyWord 校验关键字的取值规则(对应 legado getCheckKeyword):
|
||||
// 含 http/::/++/-- 的值是地址或扩展标记,不当作关键字。
|
||||
func TestSearchCheckKeyWord(t *testing.T) {
|
||||
|
||||
Reference in New Issue
Block a user