mirror of
https://github.com/tiennm99/miti99bot.git
synced 2026-10-04 14:13:26 +00:00
The monkeyd-crawler repository moved into tiennm99/mttools, so recursive clones of the old submodule URL fail and block every deploy. The crawler, PDF renderer, and export flow now live under internal/modules/monkeyd, trimmed to the bot's use: a fixed phone page, the bundled font only, and the font size as the single export option.
51 lines
1.5 KiB
Go
51 lines
1.5 KiB
Go
package crawler
|
|
|
|
import (
|
|
"regexp"
|
|
"strconv"
|
|
"strings"
|
|
)
|
|
|
|
// The site hides part of every chapter behind CSS rather than putting it in the
|
|
// markup. Chapter HTML carries empty elements such as
|
|
//
|
|
// Nghe <span class="t-3e625e..."></span> trưởng tử
|
|
//
|
|
// and the page stylesheet supplies the missing word:
|
|
//
|
|
// .t-3e625e...:before { content: "vị"; }
|
|
//
|
|
// Reading DOM text alone therefore drops hundreds of words per chapter without
|
|
// any visible error. wordRule finds those rules so the words can be put back.
|
|
var wordRule = regexp.MustCompile(`\.([A-Za-z0-9_-]+)\s*::?before\s*\{[^}]*?content\s*:\s*"((?:[^"\\]|\\.)*)"`)
|
|
|
|
// cssEscape matches a CSS character escape: a hex code point, optionally
|
|
// followed by one whitespace terminator, or an escaped literal character.
|
|
var cssEscape = regexp.MustCompile(`\\([0-9A-Fa-f]{1,6})\s?|\\(.)`)
|
|
|
|
// ParseWordClasses maps CSS class name to the word its :before rule injects.
|
|
func ParseWordClasses(page []byte) map[string]string {
|
|
words := make(map[string]string)
|
|
for _, m := range wordRule.FindAllSubmatch(page, -1) {
|
|
words[string(m[1])] = decodeCSSString(string(m[2]))
|
|
}
|
|
return words
|
|
}
|
|
|
|
// decodeCSSString resolves the escape sequences allowed inside a CSS string.
|
|
func decodeCSSString(s string) string {
|
|
if !strings.Contains(s, `\`) {
|
|
return s
|
|
}
|
|
return cssEscape.ReplaceAllStringFunc(s, func(esc string) string {
|
|
m := cssEscape.FindStringSubmatch(esc)
|
|
if m[1] != "" {
|
|
if cp, err := strconv.ParseInt(m[1], 16, 32); err == nil && cp > 0 {
|
|
return string(rune(cp))
|
|
}
|
|
return ""
|
|
}
|
|
return m[2]
|
|
})
|
|
}
|