mirror of
https://github.com/truewhile/MeBox.git
synced 2026-10-12 00:06:37 +08:00
376 lines
10 KiB
Go
376 lines
10 KiB
Go
package rule
|
||
|
||
import (
|
||
"encoding/binary"
|
||
"fmt"
|
||
"strings"
|
||
)
|
||
|
||
// 本文件移植 legado 的 QueryTTF:解析字体(sfnt)的 cmap / glyf / loca / maxp 表,
|
||
// 建立「Unicode 码点 → 字形」「字形 → Unicode 码点」两张表,供
|
||
// java.queryTTF / java.replaceFont 还原被字体混淆的正文。
|
||
//
|
||
// 阅读站点的常见套路是:正文用一套打乱过的字体渲染,页面上给出「错误字体」(把
|
||
// 每个码点映射到错字形)与「正确字体」(字形到真实码点的映射)。replaceFont 按
|
||
// 字形轮廓把错误字体里的字符逐个换成正确字体里同字形的码点,从而还原原文。
|
||
//
|
||
// 解析器全程做边界检查:字体字节来自书源(不可信),坏字体只应报错,不能 panic。
|
||
|
||
// queryTTFFont 一个已解析的字体。
|
||
type queryTTFFont struct {
|
||
// unicodeToGlyphID 码点 → 字形在 glyf 表里的下标。
|
||
unicodeToGlyphID map[rune]uint16
|
||
// unicodeToGlyph 码点 → 字形轮廓(用于跨字体比较字形)。
|
||
unicodeToGlyph map[rune]string
|
||
// glyphToUnicode 字形轮廓 → 码点(正确字体用来查回真实字符)。
|
||
glyphToUnicode map[string]rune
|
||
}
|
||
|
||
// sfnt 表标签。
|
||
var (
|
||
ttfTagCmap = [4]byte{'c', 'm', 'a', 'p'}
|
||
ttfTagGlyf = [4]byte{'g', 'l', 'y', 'f'}
|
||
ttfTagLoca = [4]byte{'l', 'o', 'c', 'a'}
|
||
ttfTagMaxp = [4]byte{'m', 'a', 'x', 'p'}
|
||
ttfTagHead = [4]byte{'h', 'e', 'a', 'd'}
|
||
)
|
||
|
||
// parseQueryTTFFont 解析字体字节。支持 sfnt(TTF/OTF)与 ttc 的第一套字体。
|
||
func parseQueryTTFFont(data []byte) (font *queryTTFFont, err error) {
|
||
defer func() {
|
||
if r := recover(); r != nil {
|
||
font, err = nil, fmt.Errorf("字体解析失败: %v", r)
|
||
}
|
||
}()
|
||
if len(data) < 12 {
|
||
return nil, fmt.Errorf("字体数据过短")
|
||
}
|
||
// ttc:取第一套字体的偏移。
|
||
if string(data[:4]) == "ttcf" {
|
||
if len(data) < 16 {
|
||
return nil, fmt.Errorf("ttc 头部不完整")
|
||
}
|
||
off := int(binary.BigEndian.Uint32(data[12:16]))
|
||
if off <= 0 || off >= len(data) {
|
||
return nil, fmt.Errorf("ttc 字体偏移非法")
|
||
}
|
||
data = data[off:]
|
||
}
|
||
numTables := int(binary.BigEndian.Uint16(data[4:6]))
|
||
if numTables <= 0 || 12+numTables*16 > len(data) {
|
||
return nil, fmt.Errorf("sfnt 表目录非法")
|
||
}
|
||
tables := map[[4]byte][]byte{}
|
||
for i := 0; i < numTables; i++ {
|
||
rec := data[12+i*16 : 12+i*16+16]
|
||
var tag [4]byte
|
||
copy(tag[:], rec[:4])
|
||
off := int(binary.BigEndian.Uint32(rec[8:12]))
|
||
length := int(binary.BigEndian.Uint32(rec[12:16]))
|
||
if off < 0 || length < 0 || off > len(data) {
|
||
continue
|
||
}
|
||
if off+length > len(data) {
|
||
length = len(data) - off
|
||
}
|
||
tables[tag] = data[off : off+length]
|
||
}
|
||
|
||
cmap := tables[ttfTagCmap]
|
||
if cmap == nil {
|
||
return nil, fmt.Errorf("字体缺少 cmap 表")
|
||
}
|
||
mapping, err := parseTTFCmap(cmap)
|
||
if err != nil {
|
||
return nil, err
|
||
}
|
||
f := &queryTTFFont{
|
||
unicodeToGlyphID: mapping,
|
||
unicodeToGlyph: map[rune]string{},
|
||
glyphToUnicode: map[string]rune{},
|
||
}
|
||
|
||
// 有 glyf + loca + head + maxp 才能算字形轮廓;缺了(如 CFF 字体)只保留码点映射。
|
||
head := tables[ttfTagHead]
|
||
maxp := tables[ttfTagMaxp]
|
||
loca := tables[ttfTagLoca]
|
||
glyf := tables[ttfTagGlyf]
|
||
if head == nil || maxp == nil || loca == nil || glyf == nil || len(head) < 54 {
|
||
return f, nil
|
||
}
|
||
indexToLocFormat := int16(binary.BigEndian.Uint16(head[50:52]))
|
||
numGlyphs := int(binary.BigEndian.Uint16(maxp[4:6]))
|
||
offsets, err := parseTTFLoca(loca, numGlyphs, indexToLocFormat)
|
||
if err != nil {
|
||
return f, nil
|
||
}
|
||
glyphCache := map[uint16]string{}
|
||
for cp, gid := range mapping {
|
||
outline := ttfGlyphOutline(glyf, offsets, gid, glyphCache, 0)
|
||
f.unicodeToGlyph[cp] = outline
|
||
if _, exists := f.glyphToUnicode[outline]; !exists {
|
||
f.glyphToUnicode[outline] = cp
|
||
}
|
||
}
|
||
return f, nil
|
||
}
|
||
|
||
// parseTTFCmap 解析 cmap 表,返回码点 → 字形下标。
|
||
// 支持 format 0 / 4 / 6(legado QueryTTF 同样只支持这三种)。
|
||
func parseTTFCmap(cmap []byte) (map[rune]uint16, error) {
|
||
if len(cmap) < 4 {
|
||
return nil, fmt.Errorf("cmap 表过短")
|
||
}
|
||
numTables := int(binary.BigEndian.Uint16(cmap[2:4]))
|
||
if 4+numTables*8 > len(cmap) {
|
||
return nil, fmt.Errorf("cmap 子表目录非法")
|
||
}
|
||
out := map[rune]uint16{}
|
||
for i := 0; i < numTables; i++ {
|
||
rec := cmap[4+i*8 : 4+i*8+8]
|
||
off := int(binary.BigEndian.Uint32(rec[4:8]))
|
||
if off < 0 || off+2 > len(cmap) {
|
||
continue
|
||
}
|
||
sub := cmap[off:]
|
||
switch binary.BigEndian.Uint16(sub[0:2]) {
|
||
case 0:
|
||
parseCmapFormat0(sub, out)
|
||
case 4:
|
||
parseCmapFormat4(sub, out)
|
||
case 6:
|
||
parseCmapFormat6(sub, out)
|
||
}
|
||
// 优先保留第一个子表解析到的映射;后续子表只补缺失项。
|
||
}
|
||
if len(out) == 0 {
|
||
return nil, fmt.Errorf("cmap 没有可用的 format 0/4/6 子表")
|
||
}
|
||
return out, nil
|
||
}
|
||
|
||
func parseCmapFormat0(sub []byte, out map[rune]uint16) {
|
||
if len(sub) < 262 {
|
||
return
|
||
}
|
||
length := int(binary.BigEndian.Uint16(sub[2:4]))
|
||
if length > len(sub) {
|
||
length = len(sub)
|
||
}
|
||
glyphs := sub[6:min(6+256, length)]
|
||
for i, gid := range glyphs {
|
||
if gid != 0 {
|
||
if _, exists := out[rune(i)]; !exists {
|
||
out[rune(i)] = uint16(gid)
|
||
}
|
||
}
|
||
}
|
||
}
|
||
|
||
func parseCmapFormat4(sub []byte, out map[rune]uint16) {
|
||
if len(sub) < 14 {
|
||
return
|
||
}
|
||
segCountX2 := int(binary.BigEndian.Uint16(sub[6:8]))
|
||
segCount := segCountX2 / 2
|
||
if segCount == 0 {
|
||
return
|
||
}
|
||
endBase := 14
|
||
startBase := endBase + segCountX2 + 2
|
||
deltaBase := startBase + segCountX2
|
||
rangeBase := deltaBase + segCountX2
|
||
if rangeBase+segCountX2 > len(sub) {
|
||
return
|
||
}
|
||
for i := 0; i < segCount; i++ {
|
||
end := int(binary.BigEndian.Uint16(sub[endBase+i*2:]))
|
||
start := int(binary.BigEndian.Uint16(sub[startBase+i*2:]))
|
||
delta := int16(binary.BigEndian.Uint16(sub[deltaBase+i*2:]))
|
||
rangeOffset := int(binary.BigEndian.Uint16(sub[rangeBase+i*2:]))
|
||
if start > end {
|
||
continue
|
||
}
|
||
for cp := start; cp <= end && cp <= 0xFFFF; cp++ {
|
||
if cp == 0xFFFF {
|
||
continue
|
||
}
|
||
var gid uint16
|
||
if rangeOffset == 0 {
|
||
gid = uint16(int(cp) + int(delta))
|
||
} else {
|
||
idx := rangeBase + i*2 + rangeOffset + (cp-start)*2
|
||
if idx+2 > len(sub) {
|
||
continue
|
||
}
|
||
g := binary.BigEndian.Uint16(sub[idx : idx+2])
|
||
if g == 0 {
|
||
continue
|
||
}
|
||
gid = uint16(int(g) + int(delta))
|
||
}
|
||
if gid != 0 {
|
||
if _, exists := out[rune(cp)]; !exists {
|
||
out[rune(cp)] = gid
|
||
}
|
||
}
|
||
}
|
||
}
|
||
}
|
||
|
||
func parseCmapFormat6(sub []byte, out map[rune]uint16) {
|
||
if len(sub) < 10 {
|
||
return
|
||
}
|
||
first := int(binary.BigEndian.Uint16(sub[6:8]))
|
||
count := int(binary.BigEndian.Uint16(sub[8:10]))
|
||
for i := 0; i < count; i++ {
|
||
idx := 10 + i*2
|
||
if idx+2 > len(sub) {
|
||
return
|
||
}
|
||
gid := binary.BigEndian.Uint16(sub[idx : idx+2])
|
||
if gid == 0 {
|
||
continue
|
||
}
|
||
cp := rune(first + i)
|
||
if _, exists := out[cp]; !exists {
|
||
out[cp] = gid
|
||
}
|
||
}
|
||
}
|
||
|
||
// parseTTFLoca 解析 loca 表,返回每个字形的字节区间起止。
|
||
func parseTTFLoca(loca []byte, numGlyphs int, indexToLocFormat int16) ([]int, error) {
|
||
if indexToLocFormat == 0 {
|
||
need := (numGlyphs + 1) * 2
|
||
if len(loca) < need {
|
||
return nil, fmt.Errorf("loca 表过短")
|
||
}
|
||
out := make([]int, numGlyphs+1)
|
||
for i := 0; i <= numGlyphs; i++ {
|
||
out[i] = int(binary.BigEndian.Uint16(loca[i*2:])) * 2
|
||
}
|
||
return out, nil
|
||
}
|
||
need := (numGlyphs + 1) * 4
|
||
if len(loca) < need {
|
||
return nil, fmt.Errorf("loca 表过短")
|
||
}
|
||
out := make([]int, numGlyphs+1)
|
||
for i := 0; i <= numGlyphs; i++ {
|
||
out[i] = int(binary.BigEndian.Uint32(loca[i*4:]))
|
||
}
|
||
return out, nil
|
||
}
|
||
|
||
// ttfGlyphOutline 把字形转成轮廓字符串(对应 legado QueryTTF.Glyf.toString)。
|
||
// 复合字形递归展开组件,深度上限防自引用。
|
||
func ttfGlyphOutline(glyf []byte, offsets []int, gid uint16, cache map[uint16]string, depth int) string {
|
||
if depth > 8 {
|
||
return fmt.Sprintf("glyph%d", gid)
|
||
}
|
||
if v, ok := cache[gid]; ok {
|
||
return v
|
||
}
|
||
if int(gid)+1 >= len(offsets) {
|
||
return fmt.Sprintf("glyph%d", gid)
|
||
}
|
||
start, end := offsets[gid], offsets[gid+1]
|
||
if start < 0 || end > len(glyf) || end <= start {
|
||
// 空字形(如空格):用下标本身当轮廓,保证不同码点不会互相误判。
|
||
out := fmt.Sprintf("glyph%d", gid)
|
||
cache[gid] = out
|
||
return out
|
||
}
|
||
numberOfContours := int16(binary.BigEndian.Uint16(glyf[start : start+2]))
|
||
if numberOfContours >= 0 {
|
||
out := fmt.Sprintf("simple:%d:%s", numberOfContours, ttfSimpleGlyphPoints(glyf[start:end]))
|
||
cache[gid] = out
|
||
return out
|
||
}
|
||
// 复合字形:逐组件展开。
|
||
var sb strings.Builder
|
||
sb.WriteString("composite")
|
||
pos := start + 10
|
||
for pos+4 <= end {
|
||
flags := binary.BigEndian.Uint16(glyf[pos : pos+2])
|
||
component := binary.BigEndian.Uint16(glyf[pos+2 : pos+4])
|
||
pos += 4
|
||
if flags&0x0001 != 0 { // ARG_1_AND_2_ARE_WORDS
|
||
pos += 4
|
||
} else {
|
||
pos += 2
|
||
}
|
||
switch {
|
||
case flags&0x0008 != 0: // WE_HAVE_A_SCALE
|
||
pos += 2
|
||
case flags&0x0040 != 0: // WE_HAVE_AN_X_AND_Y_SCALE
|
||
pos += 4
|
||
case flags&0x0080 != 0: // WE_HAVE_A_TWO_BY_TWO
|
||
pos += 8
|
||
}
|
||
sb.WriteString("+")
|
||
sb.WriteString(ttfGlyphOutline(glyf, offsets, component, cache, depth+1))
|
||
if flags&0x0020 == 0 { // MORE_COMPONENTS
|
||
break
|
||
}
|
||
}
|
||
out := sb.String()
|
||
cache[gid] = out
|
||
return out
|
||
}
|
||
|
||
// ttfSimpleGlyphPoints 取简单字形的轮廓点(标志与坐标的紧凑编码)。
|
||
func ttfSimpleGlyphPoints(data []byte) string {
|
||
if len(data) < 10 {
|
||
return ""
|
||
}
|
||
numberOfContours := int(binary.BigEndian.Uint16(data[0:2]))
|
||
if numberOfContours <= 0 {
|
||
return ""
|
||
}
|
||
endPtsPos := 10
|
||
if endPtsPos+numberOfContours*2+2 > len(data) {
|
||
return ""
|
||
}
|
||
numPoints := int(binary.BigEndian.Uint16(data[endPtsPos+(numberOfContours-1)*2:])) + 1
|
||
if numPoints <= 0 {
|
||
return ""
|
||
}
|
||
pos := endPtsPos + numberOfContours*2 + 2 // + instructionLength
|
||
if pos > len(data) {
|
||
return ""
|
||
}
|
||
instrLen := int(binary.BigEndian.Uint16(data[pos-2 : pos]))
|
||
pos += instrLen
|
||
flags := make([]byte, 0, numPoints)
|
||
for len(flags) < numPoints && pos < len(data) {
|
||
flag := data[pos]
|
||
pos++
|
||
flags = append(flags, flag)
|
||
if flag&0x08 != 0 { // REPEAT
|
||
if pos >= len(data) {
|
||
break
|
||
}
|
||
repeat := int(data[pos])
|
||
pos++
|
||
for i := 0; i < repeat && len(flags) < numPoints; i++ {
|
||
flags = append(flags, flag)
|
||
}
|
||
}
|
||
}
|
||
// 解析 x / y 坐标(与点数等长的增量序列,这里只用于区分字形)。
|
||
var sb strings.Builder
|
||
sb.WriteString(fmt.Sprintf("n=%d;", numPoints))
|
||
for _, flag := range flags {
|
||
sb.WriteByte('0' + flag&0x0F)
|
||
}
|
||
sb.WriteString(";")
|
||
// 跳过坐标数据不影响「同名轮廓一致性」的判断:同字形的字体坐标编码一致。
|
||
if pos < len(data) {
|
||
sb.WriteString(fmt.Sprintf("d=%d", len(data)-pos))
|
||
}
|
||
return sb.String()
|
||
}
|