mirror of
https://github.com/truewhile/MeBox.git
synced 2026-10-08 14:26:37 +08:00
feat(reader): legado 书源规则引擎 Go 移植 + 阅读 P0 后端
- internal/service/reader/rule/:逐方法移植 legado analyzeRule 包 (AnalyzeRule/AnalyzeByJSoup/AnalyzeByJSonPath/AnalyzeByXPath/ AnalyzeByRegex/AnalyzeUrl/RuleAnalyzer),JS 规则留 P2 接入点 - 书源导入(JSON数组/对象/Base64/URL)、多源聚合搜索(legado 四档 排序合并)、详情/目录/正文(nextContentUrl 翻页合并) - 数据模型 4 表注册迁移;/api/reader/* 路由组(书源/搜索/书架/进度/调试) - 单测 + httptest 全链路端到端测试 - docs/reader-ui-spec.md:legado UI 交互仿制规格(供 P1 前端使用)
This commit is contained in:
@@ -0,0 +1,416 @@
|
||||
package rule
|
||||
|
||||
import "strings"
|
||||
|
||||
// RuleAnalyzer 移植自 legado RuleAnalyzer.kt,逐方法对齐。
|
||||
// 用于按 "@"/"&&"/"||"/"%%"/"@" 切分规则字符串,切分时跳过引号内与
|
||||
// 平衡组([...]、(...))内的分隔符,避免与选择器或正则内容冲突。
|
||||
|
||||
// RuleAnalyzer 是无状态的切分器实例(一次使用)。
|
||||
type RuleAnalyzer struct {
|
||||
queue string
|
||||
pos int
|
||||
start int
|
||||
startX int
|
||||
rule []string
|
||||
step int
|
||||
elementsType string
|
||||
code bool
|
||||
}
|
||||
|
||||
// NewRuleAnalyzer 对应 RuleAnalyzer(data, code);code=true 时平衡组按
|
||||
// 代码语义(处理转义、区分 [] 与 ())处理,用于 jsonPath 规则。
|
||||
func NewRuleAnalyzer(data string, code bool) *RuleAnalyzer {
|
||||
return &RuleAnalyzer{queue: data, code: code}
|
||||
}
|
||||
|
||||
// ElementsType 返回上次 splitRule 使用的组合符("&&"/"||"/"%%")。
|
||||
func (a *RuleAnalyzer) ElementsType() string { return a.elementsType }
|
||||
|
||||
// Rules 返回切分结果。
|
||||
func (a *RuleAnalyzer) Rules() []string { return a.rule }
|
||||
|
||||
// Trim 对应 trim():修剪当前规则之前的 "@" 或不可见字符。
|
||||
func (a *RuleAnalyzer) Trim() {
|
||||
if a.pos >= len(a.queue) {
|
||||
return
|
||||
}
|
||||
if a.queue[a.pos] == '@' || a.queue[a.pos] < '!' {
|
||||
a.pos++
|
||||
for a.pos < len(a.queue) && (a.queue[a.pos] == '@' || a.queue[a.pos] < '!') {
|
||||
a.pos++
|
||||
}
|
||||
a.start = a.pos
|
||||
a.startX = a.pos
|
||||
}
|
||||
}
|
||||
|
||||
// ReSetPos 对应 reSetPos()。
|
||||
func (a *RuleAnalyzer) ReSetPos() {
|
||||
a.pos = 0
|
||||
a.startX = 0
|
||||
}
|
||||
|
||||
// consumeTo 对应 consumeTo(seq)。
|
||||
func (a *RuleAnalyzer) consumeTo(seq string) bool {
|
||||
a.start = a.pos
|
||||
offset := strings.Index(a.queue[a.pos:], seq)
|
||||
if offset == -1 {
|
||||
return false
|
||||
}
|
||||
a.pos += offset
|
||||
return true
|
||||
}
|
||||
|
||||
// consumeToAny 对应 consumeToAny(seq)。
|
||||
func (a *RuleAnalyzer) consumeToAny(seqs ...string) bool {
|
||||
pos := a.pos
|
||||
for pos != len(a.queue) {
|
||||
for _, s := range seqs {
|
||||
if strings.HasPrefix(a.queue[pos:], s) {
|
||||
a.step = len(s)
|
||||
a.pos = pos
|
||||
return true
|
||||
}
|
||||
}
|
||||
pos++
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// findToAny 对应 findToAny(seq: Char)。
|
||||
func (a *RuleAnalyzer) findToAny(chars ...byte) int {
|
||||
pos := a.pos
|
||||
for pos != len(a.queue) {
|
||||
for _, c := range chars {
|
||||
if a.queue[pos] == c {
|
||||
return pos
|
||||
}
|
||||
}
|
||||
pos++
|
||||
}
|
||||
return -1
|
||||
}
|
||||
|
||||
// chompCodeBalanced 对应 chompCodeBalanced(open, close):拉出一个平衡组,
|
||||
// 存在转义文本,'[' 与参数对 (open/close) 分别计数。
|
||||
func (a *RuleAnalyzer) chompCodeBalanced(open, close byte) bool {
|
||||
pos := a.pos
|
||||
depth := 0
|
||||
otherDepth := 0
|
||||
inSingle := false
|
||||
inDouble := false
|
||||
for {
|
||||
if pos == len(a.queue) {
|
||||
break
|
||||
}
|
||||
c := a.queue[pos]
|
||||
pos++
|
||||
if c != '\\' {
|
||||
if c == '\'' && !inDouble {
|
||||
inSingle = !inSingle
|
||||
} else if c == '"' && !inSingle {
|
||||
inDouble = !inDouble
|
||||
}
|
||||
if inSingle || inDouble {
|
||||
continue
|
||||
}
|
||||
if c == '[' {
|
||||
depth++
|
||||
} else if c == ']' {
|
||||
depth--
|
||||
} else if depth == 0 {
|
||||
if c == open {
|
||||
otherDepth++
|
||||
} else if c == close {
|
||||
otherDepth--
|
||||
}
|
||||
}
|
||||
} else {
|
||||
pos++
|
||||
}
|
||||
if !(depth > 0 || otherDepth > 0) {
|
||||
break
|
||||
}
|
||||
}
|
||||
if depth > 0 || otherDepth > 0 {
|
||||
return false
|
||||
}
|
||||
a.pos = pos
|
||||
return true
|
||||
}
|
||||
|
||||
// chompRuleBalanced 对应 chompRuleBalanced(open, close):引号外才处理转义。
|
||||
func (a *RuleAnalyzer) chompRuleBalanced(open, close byte) bool {
|
||||
pos := a.pos
|
||||
depth := 0
|
||||
inSingle := false
|
||||
inDouble := false
|
||||
for {
|
||||
if pos == len(a.queue) {
|
||||
break
|
||||
}
|
||||
c := a.queue[pos]
|
||||
pos++
|
||||
if c == '\'' && !inDouble {
|
||||
inSingle = !inSingle
|
||||
} else if c == '"' && !inSingle {
|
||||
inDouble = !inDouble
|
||||
}
|
||||
if inSingle || inDouble {
|
||||
continue
|
||||
}
|
||||
if c == '\\' {
|
||||
pos++
|
||||
continue
|
||||
}
|
||||
if c == open {
|
||||
depth++
|
||||
} else if c == close {
|
||||
depth--
|
||||
}
|
||||
if !(depth > 0) {
|
||||
break
|
||||
}
|
||||
}
|
||||
if depth > 0 {
|
||||
return false
|
||||
}
|
||||
a.pos = pos
|
||||
return true
|
||||
}
|
||||
|
||||
func (a *RuleAnalyzer) chompBalanced(open, close byte) bool {
|
||||
if a.code {
|
||||
return a.chompCodeBalanced(open, close)
|
||||
}
|
||||
return a.chompRuleBalanced(open, close)
|
||||
}
|
||||
|
||||
// SplitRule 对应 splitRule(vararg split):把 queue 切分成规则列表。
|
||||
// 首段按 splits 中最先出现的分隔符确定组合类型,其余按该类型循环切分。
|
||||
func (a *RuleAnalyzer) SplitRule(splits ...string) []string {
|
||||
if len(splits) == 1 {
|
||||
a.elementsType = splits[0]
|
||||
if !a.consumeTo(a.elementsType) {
|
||||
a.rule = append(a.rule, a.queue[a.startX:])
|
||||
return a.rule
|
||||
}
|
||||
a.step = len(a.elementsType)
|
||||
return a.splitRuleNext()
|
||||
}
|
||||
if !a.consumeToAny(splits...) {
|
||||
a.rule = append(a.rule, a.queue[a.startX:])
|
||||
return a.rule
|
||||
}
|
||||
|
||||
end := a.pos
|
||||
a.pos = a.start
|
||||
for {
|
||||
st := a.findToAny('[', '(')
|
||||
if st == -1 {
|
||||
a.rule = []string{a.queue[a.startX:end]}
|
||||
a.elementsType = a.queue[end : end+a.step]
|
||||
a.pos = end + a.step
|
||||
for a.consumeTo(a.elementsType) {
|
||||
a.rule = append(a.rule, a.queue[a.start:a.pos])
|
||||
a.pos += a.step
|
||||
}
|
||||
a.rule = append(a.rule, a.queue[a.pos:])
|
||||
return a.rule
|
||||
}
|
||||
if st > end {
|
||||
a.rule = []string{a.queue[a.startX:end]}
|
||||
a.elementsType = a.queue[end : end+a.step]
|
||||
a.pos = end + a.step
|
||||
for a.consumeTo(a.elementsType) && a.pos < st {
|
||||
a.rule = append(a.rule, a.queue[a.start:a.pos])
|
||||
a.pos += a.step
|
||||
}
|
||||
if a.pos > st {
|
||||
a.startX = a.start
|
||||
return a.splitRuleNext()
|
||||
}
|
||||
a.rule = append(a.rule, a.queue[a.pos:])
|
||||
return a.rule
|
||||
}
|
||||
a.pos = st
|
||||
next := byte(')')
|
||||
if a.queue[a.pos] == '[' {
|
||||
next = ']'
|
||||
}
|
||||
if !a.chompBalanced(a.queue[a.pos], next) {
|
||||
return []string{a.queue}
|
||||
}
|
||||
if end <= a.pos {
|
||||
break
|
||||
}
|
||||
}
|
||||
a.start = a.pos
|
||||
return a.splitRuleFirst(splits...)
|
||||
}
|
||||
|
||||
// splitRuleFirst 对应首段匹配的 tailrec 递归(elementsType 尚未确定)。
|
||||
func (a *RuleAnalyzer) splitRuleFirst(splits ...string) []string {
|
||||
if len(splits) == 1 {
|
||||
a.elementsType = splits[0]
|
||||
if !a.consumeTo(a.elementsType) {
|
||||
a.rule = append(a.rule, a.queue[a.startX:])
|
||||
return a.rule
|
||||
}
|
||||
a.step = len(a.elementsType)
|
||||
return a.splitRuleNext()
|
||||
}
|
||||
if !a.consumeToAny(splits...) {
|
||||
a.rule = append(a.rule, a.queue[a.startX:])
|
||||
return a.rule
|
||||
}
|
||||
end := a.pos
|
||||
a.pos = a.start
|
||||
for {
|
||||
st := a.findToAny('[', '(')
|
||||
if st == -1 {
|
||||
a.rule = []string{a.queue[a.startX:end]}
|
||||
a.elementsType = a.queue[end : end+a.step]
|
||||
a.pos = end + a.step
|
||||
for a.consumeTo(a.elementsType) {
|
||||
a.rule = append(a.rule, a.queue[a.start:a.pos])
|
||||
a.pos += a.step
|
||||
}
|
||||
a.rule = append(a.rule, a.queue[a.pos:])
|
||||
return a.rule
|
||||
}
|
||||
if st > end {
|
||||
a.rule = []string{a.queue[a.startX:end]}
|
||||
a.elementsType = a.queue[end : end+a.step]
|
||||
a.pos = end + a.step
|
||||
for a.consumeTo(a.elementsType) && a.pos < st {
|
||||
a.rule = append(a.rule, a.queue[a.start:a.pos])
|
||||
a.pos += a.step
|
||||
}
|
||||
if a.pos > st {
|
||||
a.startX = a.start
|
||||
return a.splitRuleNext()
|
||||
}
|
||||
a.rule = append(a.rule, a.queue[a.pos:])
|
||||
return a.rule
|
||||
}
|
||||
a.pos = st
|
||||
next := byte(')')
|
||||
if a.queue[a.pos] == '[' {
|
||||
next = ']'
|
||||
}
|
||||
if !a.chompBalanced(a.queue[a.pos], next) {
|
||||
return []string{a.queue}
|
||||
}
|
||||
if end <= a.pos {
|
||||
break
|
||||
}
|
||||
}
|
||||
a.start = a.pos
|
||||
return a.splitRuleFirst(splits...)
|
||||
}
|
||||
|
||||
// splitRuleNext 对应二段匹配 splitRule()(elementsType 已确定)。
|
||||
func (a *RuleAnalyzer) splitRuleNext() []string {
|
||||
for {
|
||||
end := a.pos
|
||||
a.pos = a.start
|
||||
for {
|
||||
st := a.findToAny('[', '(')
|
||||
if st == -1 {
|
||||
a.rule = append(a.rule, a.queue[a.startX:end])
|
||||
a.pos = end + a.step
|
||||
for a.consumeTo(a.elementsType) {
|
||||
a.rule = append(a.rule, a.queue[a.start:a.pos])
|
||||
a.pos += a.step
|
||||
}
|
||||
a.rule = append(a.rule, a.queue[a.pos:])
|
||||
return a.rule
|
||||
}
|
||||
if st > end {
|
||||
a.rule = append(a.rule, a.queue[a.startX:end])
|
||||
a.pos = end + a.step
|
||||
for a.consumeTo(a.elementsType) && a.pos < st {
|
||||
a.rule = append(a.rule, a.queue[a.start:a.pos])
|
||||
a.pos += a.step
|
||||
}
|
||||
if a.pos > st {
|
||||
a.startX = a.start
|
||||
return a.splitRuleNext()
|
||||
}
|
||||
a.rule = append(a.rule, a.queue[a.pos:])
|
||||
return a.rule
|
||||
}
|
||||
a.pos = st
|
||||
next := byte(')')
|
||||
if a.queue[a.pos] == '[' {
|
||||
next = ']'
|
||||
}
|
||||
if !a.chompBalanced(a.queue[a.pos], next) {
|
||||
return []string{a.queue}
|
||||
}
|
||||
if end <= a.pos {
|
||||
break
|
||||
}
|
||||
}
|
||||
a.start = a.pos
|
||||
if !a.consumeTo(a.elementsType) {
|
||||
a.rule = append(a.rule, a.queue[a.startX:])
|
||||
return a.rule
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// InnerRule 对应 innerRule(inner, startStep, endStep, fr) 第一变体:
|
||||
// 替换所有内嵌规则(如 {$.xxx}),fr 返回空表示该处不是有效内嵌规则。
|
||||
// 返回替换后的完整字符串;无一替换成功时返回 ""。
|
||||
func (a *RuleAnalyzer) InnerRule(inner string, startStep, endStep int, fr func(string) string) string {
|
||||
var sb []byte
|
||||
for {
|
||||
if !a.consumeTo(inner) {
|
||||
break
|
||||
}
|
||||
posPre := a.pos
|
||||
if a.chompCodeBalanced('{', '}') {
|
||||
frv := fr(a.queue[posPre+startStep : a.pos-endStep])
|
||||
if frv != "" {
|
||||
sb = append(sb, a.queue[a.startX:posPre]...)
|
||||
sb = append(sb, frv...)
|
||||
a.startX = a.pos
|
||||
continue
|
||||
}
|
||||
}
|
||||
a.pos += len(inner)
|
||||
}
|
||||
if a.startX == 0 {
|
||||
return ""
|
||||
}
|
||||
sb = append(sb, a.queue[a.startX:]...)
|
||||
return string(sb)
|
||||
}
|
||||
|
||||
// InnerRule2 对应 innerRule(startStr, endStr, fr) 第二变体:无平衡组检查。
|
||||
// 无一替换成功时返回原串。
|
||||
func (a *RuleAnalyzer) InnerRule2(startStr, endStr string, fr func(string) string) string {
|
||||
var sb []byte
|
||||
for {
|
||||
if !a.consumeTo(startStr) {
|
||||
break
|
||||
}
|
||||
a.pos += len(startStr)
|
||||
posPre := a.pos
|
||||
if a.consumeTo(endStr) {
|
||||
frv := fr(a.queue[posPre:a.pos])
|
||||
sb = append(sb, a.queue[a.startX:posPre-len(startStr)]...)
|
||||
sb = append(sb, frv...)
|
||||
a.pos += len(endStr)
|
||||
a.startX = a.pos
|
||||
}
|
||||
}
|
||||
if a.startX == 0 {
|
||||
return a.queue
|
||||
}
|
||||
sb = append(sb, a.queue[a.startX:]...)
|
||||
return string(sb)
|
||||
}
|
||||
Reference in New Issue
Block a user