mirror of
https://github.com/truewhile/MeBox.git
synced 2026-10-01 20:16:36 +08:00
b8e6418326
- internal/service/reader/rule/:逐方法移植 legado analyzeRule 包 (AnalyzeRule/AnalyzeByJSoup/AnalyzeByJSonPath/AnalyzeByXPath/ AnalyzeByRegex/AnalyzeUrl/RuleAnalyzer),JS 规则留 P2 接入点 - 书源导入(JSON数组/对象/Base64/URL)、多源聚合搜索(legado 四档 排序合并)、详情/目录/正文(nextContentUrl 翻页合并) - 数据模型 4 表注册迁移;/api/reader/* 路由组(书源/搜索/书架/进度/调试) - 单测 + httptest 全链路端到端测试 - docs/reader-ui-spec.md:legado UI 交互仿制规格(供 P1 前端使用)
150 lines
3.8 KiB
Go
150 lines
3.8 KiB
Go
package rule
|
||
|
||
import (
|
||
"strings"
|
||
|
||
"github.com/dlclark/regexp2"
|
||
)
|
||
|
||
// 本文件对应 AnalyzeByRegex.kt。Java 正则语义用 regexp2 对齐
|
||
// (支持前向后向断言与反向引用),匹配循环对齐 Matcher.find()。
|
||
|
||
// splitNotBlankAndTrim 对应 String.splitNotBlank("&&"):切分并去空白项。
|
||
func splitNotBlankAndTrim(s, sep string) []string {
|
||
var out []string
|
||
for _, p := range strings.Split(s, sep) {
|
||
if t := strings.TrimSpace(p); t != "" {
|
||
out = append(out, t)
|
||
}
|
||
}
|
||
return out
|
||
}
|
||
|
||
// regexGetElement 对应 AnalyzeByRegex.getElement:多段正则串联,
|
||
// 最终返回第一个匹配的全部分组(含 group 0)。
|
||
func regexGetElement(res string, regs []string, index int) []string {
|
||
if index >= len(regs) {
|
||
return nil
|
||
}
|
||
re, err := regexp2.Compile(regs[index], regexp2.None)
|
||
if err != nil {
|
||
return nil
|
||
}
|
||
m, err := re.FindStringMatchStartingAt(res, 0)
|
||
if err != nil || m == nil {
|
||
return nil
|
||
}
|
||
if index+1 == len(regs) {
|
||
info := make([]string, 0, len(m.Groups()))
|
||
for _, g := range m.Groups() {
|
||
if len(g.Captures) > 0 {
|
||
info = append(info, g.Captures[0].String())
|
||
} else {
|
||
info = append(info, "")
|
||
}
|
||
}
|
||
return info
|
||
}
|
||
var sb strings.Builder
|
||
for m != nil {
|
||
sb.WriteString(m.String())
|
||
m, _ = re.FindNextMatch(m)
|
||
}
|
||
return regexGetElement(sb.String(), regs, index+1)
|
||
}
|
||
|
||
// regexGetElements 对应 AnalyzeByRegex.getElements:多段正则串联,
|
||
// 最终按每个匹配返回一组分组列表。
|
||
func regexGetElements(res string, regs []string, index int) [][]string {
|
||
if index >= len(regs) {
|
||
return nil
|
||
}
|
||
re, err := regexp2.Compile(regs[index], regexp2.None)
|
||
if err != nil {
|
||
return nil
|
||
}
|
||
m, err := re.FindStringMatchStartingAt(res, 0)
|
||
if err != nil || m == nil {
|
||
return nil
|
||
}
|
||
if index+1 == len(regs) {
|
||
var books [][]string
|
||
for m != nil {
|
||
info := make([]string, 0, len(m.Groups()))
|
||
for _, g := range m.Groups() {
|
||
if len(g.Captures) > 0 {
|
||
info = append(info, g.Captures[0].String())
|
||
} else {
|
||
info = append(info, "")
|
||
}
|
||
}
|
||
books = append(books, info)
|
||
m, _ = re.FindNextMatch(m)
|
||
}
|
||
return books
|
||
}
|
||
var sb strings.Builder
|
||
for m != nil {
|
||
sb.WriteString(m.String())
|
||
m, _ = re.FindNextMatch(m)
|
||
}
|
||
return regexGetElements(sb.String(), regs, index+1)
|
||
}
|
||
|
||
// regexReplaceAll 对应 Kotlin Regex.replace(result, replacement)
|
||
// (Java $N 分组替换语义,regexp2 的 Replace 原生支持)。
|
||
func regexReplaceAll(pattern, result, replacement string) string {
|
||
re, err := regexp2.Compile(pattern, regexp2.None)
|
||
if err != nil {
|
||
return strings.ReplaceAll(result, pattern, replacement)
|
||
}
|
||
out, err := re.Replace(result, replacement, 0, -1)
|
||
if err != nil {
|
||
return result
|
||
}
|
||
return out
|
||
}
|
||
|
||
// ApplyReplaceRegexString 应用书源 replaceRegex 字符串("##pattern##replace[##x]" 格式),
|
||
// 对应 ContentRule.replaceRegex 的处理。
|
||
func ApplyReplaceRegexString(content, replaceRegex string) string {
|
||
if replaceRegex == "" {
|
||
return content
|
||
}
|
||
segs := strings.Split(replaceRegex, "##")
|
||
if len(segs) < 2 {
|
||
return content
|
||
}
|
||
pattern := segs[1]
|
||
replacement := ""
|
||
replaceFirst := false
|
||
if len(segs) > 2 {
|
||
replacement = segs[2]
|
||
}
|
||
if len(segs) > 3 {
|
||
replaceFirst = true
|
||
}
|
||
if replaceFirst {
|
||
return regexReplaceFirstOnFirstMatch(pattern, content, replacement)
|
||
}
|
||
return regexReplaceAll(pattern, content, replacement)
|
||
}
|
||
|
||
// regexReplaceFirstOnFirstMatch 对应 replaceRegex 的 replaceFirst 分支:
|
||
// 找到第一个匹配(无匹配返回 ""),在匹配文本上做首次替换。
|
||
func regexReplaceFirstOnFirstMatch(pattern, result, replacement string) string {
|
||
re, err := regexp2.Compile(pattern, regexp2.None)
|
||
if err != nil {
|
||
return replacement
|
||
}
|
||
m, err := re.FindStringMatch(result)
|
||
if err != nil || m == nil {
|
||
return ""
|
||
}
|
||
out, err := re.Replace(m.String(), replacement, 0, 1)
|
||
if err != nil {
|
||
return m.String()
|
||
}
|
||
return out
|
||
}
|