refactor(monkeyd): port the crawler in-tree and drop the submodule

The monkeyd-crawler repository moved into tiennm99/mttools, so recursive
clones of the old submodule URL fail and block every deploy. The crawler,
PDF renderer, and export flow now live under internal/modules/monkeyd,
trimmed to the bot's use: a fixed phone page, the bundled font only, and
the font size as the single export option.
This commit is contained in:
tiennm99 committed 2026-10-03 11:23:45 +07:00
1 parent c1393f0ec8
commit d211124aae
26 files changed
+1818 -30

No files matched your search

-5
View File
@@ -16,12 +16,7 @@ jobs:
matrix:
go: ['1.26.5']
steps:
# The monkeyd module builds against third_party/monkeyd-crawler, which is
# a submodule wired in through a go.mod replace directive. Without it
# checked out, every Go step fails to resolve the package.
- uses: actions/checkout@v6
with:
submodules: true
- uses: actions/setup-go@v6
with:
-3
View File
@@ -1,3 +0,0 @@
[submodule "third_party/monkeyd-crawler"]
path = third_party/monkeyd-crawler
url = https://github.com/tiennm99/monkeyd-crawler.git
+3
View File
@@ -50,6 +50,9 @@ linters:
- name: unused-parameter
disabled: true
exclusions:
# renderer/ is the Node service; its node_modules can carry stray Go files.
paths:
- renderer
rules:
# Tests routinely pass dummy values, swallow errors from helpers, and
# use higher cyclomatic complexity in table-driven cases. Suppress the
+2 -5
View File
@@ -1,12 +1,9 @@
FROM golang:1.26.5-alpine AS builder
WORKDIR /src
# The monkeyd-crawler submodule is resolved through a `replace` directive, so
# its go.mod must be present before `go mod download` can read the build list.
# Only the module files are copied here, keeping this layer cached across
# ordinary source edits.
# Only the module files are copied first, keeping the download layer cached
# across ordinary source edits.
COPY go.mod go.sum ./
COPY third_party/monkeyd-crawler/go.mod third_party/monkeyd-crawler/go.sum ./third_party/monkeyd-crawler/
RUN go mod download
COPY . .
+3 -9
View File
@@ -3,22 +3,19 @@ module github.com/tiennm99/miti99bot
go 1.26.5
require (
github.com/go-pdf/fpdf v0.9.0
github.com/go-telegram/bot v1.20.0
github.com/ledongthuc/pdf v0.0.0-20260907135840-6c8c28e0e8a0
github.com/robfig/cron/v3 v3.0.1
github.com/testcontainers/testcontainers-go v0.43.0
github.com/testcontainers/testcontainers-go/modules/mongodb v0.43.0
github.com/tiennm99/monkeyd-crawler v0.0.0
go.mongodb.org/mongo-driver/v2 v2.7.0
golang.org/x/image v0.45.0
golang.org/x/net v0.57.0
golang.org/x/sync v0.22.0
golang.org/x/text v0.41.0
)
require (
github.com/go-pdf/fpdf v0.9.0 // indirect
golang.org/x/net v0.57.0 // indirect
)
require (
dario.cat/mergo v1.0.2 // indirect
github.com/Azure/go-ansiterm v0.0.0-20250102033503-faa5f7b0171c // indirect
@@ -72,9 +69,6 @@ require (
go.opentelemetry.io/otel/metric v1.41.0 // indirect
go.opentelemetry.io/otel/trace v1.41.0 // indirect
golang.org/x/crypto v0.54.0 // indirect
golang.org/x/sync v0.22.0 // indirect
golang.org/x/sys v0.47.0 // indirect
gopkg.in/yaml.v3 v3.0.1 // indirect
)
replace github.com/tiennm99/monkeyd-crawler => ./third_party/monkeyd-crawler
+187
View File
@@ -0,0 +1,187 @@
package crawler
import (
"bytes"
"fmt"
"strconv"
"strings"
"golang.org/x/net/html"
)
// contentElementID is the container holding a chapter's body text.
const contentElementID = "chapter-content-render"
// Chapter is one fetched chapter, reduced to plain paragraphs.
type Chapter struct {
Label string
URL string
Paragraphs []string
}
// Heading is the chapter title to print. Labels are often a bare number, which
// reads poorly as a heading, so those get the Vietnamese word for "chapter".
func (c *Chapter) Heading() string {
label := strings.TrimSpace(c.Label)
if label == "" {
return "Chương"
}
if _, err := strconv.Atoi(label); err == nil {
return "Chương " + label
}
return label
}
// WordCount is a rough word count, used to sanity-check extraction.
func (c *Chapter) WordCount() int {
n := 0
for _, p := range c.Paragraphs {
n += len(strings.Fields(p))
}
return n
}
// blockTags end the current paragraph when opened or closed.
var blockTags = map[string]bool{
"p": true, "div": true, "br": true, "hr": true, "blockquote": true,
"h1": true, "h2": true, "h3": true, "h4": true, "h5": true, "h6": true,
"li": true, "ul": true, "ol": true, "tr": true,
}
// skipTags never contribute prose.
//
// Anchors are included because inside a chapter body they are always site
// chrome: the prev/next chapter navigation and the sponsor call-to-action are
// both links, while novel prose never needs one. Inline emphasis tags (b, i,
// em) are deliberately absent so italics in the prose survive; the site's icons
// use <i> but carry no text.
var skipTags = map[string]bool{
"script": true, "style": true, "noscript": true, "iframe": true,
"ins": true, "form": true, "select": true, "button": true, "textarea": true,
"a": true, "img": true, "svg": true,
}
// junkClasses marks containers the site injects into the chapter body. Their
// whole subtree is dropped.
//
// These are matched on class rather than position because the blocks move: the
// sponsor block opens the body on most chapters but is absent on others, and
// the watermark is planted at a different paragraph in every chapter. Only
// site-specific class names are listed; generic Bootstrap utilities such as
// "my-4" or "text-center" are not, since prose could legitimately carry them.
//
// Note the sibling class "actac" is NOT junk: it wraps the real chapter text and
// carries style="display:none", because the site gates the body behind a click
// on the sponsor link and reveals it with JavaScript. Skipping hidden elements,
// or skipping "act*" as a family, would therefore discard the whole chapter.
var junkClasses = map[string]bool{
"actcl": true, // sponsor block shown in place of the gated chapter body
"signature": true, // "[Truyện được đăng tải duy nhất tại ...]" source watermark
}
// hasJunkClass reports whether a node is an injected non-prose container.
func hasJunkClass(n *html.Node) bool {
for _, tok := range strings.Fields(attr(n, "class")) {
if junkClasses[tok] {
return true
}
}
return false
}
// ParseChapter extracts a chapter's paragraphs, restoring the words the site
// serves through CSS :before rules instead of markup.
func ParseChapter(page []byte, ref ChapterRef) (*Chapter, error) {
doc, err := html.Parse(bytes.NewReader(page))
if err != nil {
return nil, fmt.Errorf("parse chapter %s: %w", ref.URL, err)
}
content := elementByID(doc, contentElementID)
if content == nil {
return nil, fmt.Errorf("chapter %s: no #%s container (page layout may have changed)",
ref.URL, contentElementID)
}
ch := &Chapter{
Label: ref.Label,
URL: ref.URL,
Paragraphs: extractParagraphs(content, ParseWordClasses(page)),
}
if len(ch.Paragraphs) == 0 {
return nil, fmt.Errorf("chapter %s: extracted no text", ref.URL)
}
ch.Paragraphs = dropRepeatedTitle(ch.Paragraphs, ref.Label)
return ch, nil
}
// extractParagraphs walks the content subtree into plain paragraphs, replacing
// each word-carrying element with the word its CSS rule injects.
func extractParagraphs(content *html.Node, words map[string]string) []string {
var b strings.Builder
var walk func(*html.Node)
walk = func(n *html.Node) {
switch n.Type {
case html.TextNode:
// The HTML parser has already decoded entities such as &#432;.
b.WriteString(n.Data)
return
case html.ElementNode:
if skipTags[n.Data] || hasJunkClass(n) {
return
}
// These elements are empty in the markup; the CSS word replaces them.
if word, ok := injectedWord(n, words); ok {
b.WriteString(word)
return
}
if blockTags[n.Data] {
b.WriteByte('\n')
}
}
for c := n.FirstChild; c != nil; c = c.NextSibling {
walk(c)
}
if n.Type == html.ElementNode && blockTags[n.Data] {
b.WriteByte('\n')
}
}
walk(content)
var paragraphs []string
for _, line := range strings.Split(b.String(), "\n") {
if p := collapseSpaces(line); p != "" {
paragraphs = append(paragraphs, p)
}
}
return paragraphs
}
// injectedWord returns the word a node's class supplies via CSS, if any.
func injectedWord(n *html.Node, words map[string]string) (string, bool) {
class := attr(n, "class")
if class == "" {
return "", false
}
for _, tok := range strings.Fields(class) {
if word, ok := words[tok]; ok {
return word, true
}
}
return "", false
}
// dropRepeatedTitle removes a leading paragraph that only repeats the chapter
// label, since the export prints its own heading.
func dropRepeatedTitle(paragraphs []string, label string) []string {
if len(paragraphs) < 2 {
return paragraphs
}
first := strings.TrimSpace(paragraphs[0])
if strings.EqualFold(first, strings.TrimSpace(label)) {
return paragraphs[1:]
}
return paragraphs
}
@@ -0,0 +1,200 @@
package crawler
import (
"strings"
"testing"
)
// chapterFixture mirrors the real page shape: words split between markup and
// CSS :before rules, HTML entities, &nbsp; spacer paragraphs and an ad script
// inside the content container.
const chapterFixture = `<!DOCTYPE html><html><head>
<style>
.t-aaa:before { content: "v\1ecb "; }
.j-bbb:before { content: "tr\1ed3 "; }
.z-ccc::before{content:"n\E0ng";}
.unused-ddd:before { content: "khong-dung"; }
</style></head><body>
<h1 class="card-title">TEN TRUYEN - 1</h1>
<div class="content-container" id="chapter-content-render">
<p>1</p>
<p>&nbsp;</p>
<p>Nghe <span class="t-aaa"></span> tr&#432;&#7903;ng t&#7917; noi.</p>
<p>&nbsp;</p>
<p>M&#7897;t <span class="j-bbb"></span> v&#224; <span class="z-ccc"></span> di.</p>
<script>ads();</script>
</div></body></html>`
func TestParseWordClassesDecodesEscapes(t *testing.T) {
words := ParseWordClasses([]byte(chapterFixture))
for class, want := range map[string]string{
"t-aaa": "vị",
"j-bbb": "trồ",
"z-ccc": "nàng",
} {
if got := words[class]; got != want {
t.Errorf("words[%q] = %q, want %q", class, got, want)
}
}
if len(words) != 4 {
t.Errorf("got %d rules, want 4", len(words))
}
}
func TestParseChapterRestoresCSSWords(t *testing.T) {
ch, err := ParseChapter([]byte(chapterFixture), ChapterRef{Label: "1", URL: "http://x/1.html"})
if err != nil {
t.Fatalf("ParseChapter: %v", err)
}
want := []string{
"Nghe vị trưởng tử noi.",
"Một trồ và nàng di.",
}
if len(ch.Paragraphs) != len(want) {
t.Fatalf("got %d paragraphs %q, want %d", len(ch.Paragraphs), ch.Paragraphs, len(want))
}
for i, w := range want {
if ch.Paragraphs[i] != w {
t.Errorf("paragraph %d = %q, want %q", i, ch.Paragraphs[i], w)
}
}
}
// The CSS-injected words are the difference between real text and text with
// silent holes, so guard against a regression that drops them.
func TestParseChapterWithoutCSSWouldLoseWords(t *testing.T) {
withoutCSS := strings.Replace(chapterFixture, `.t-aaa:before { content: "v\1ecb "; }`, "", 1)
ch, err := ParseChapter([]byte(withoutCSS), ChapterRef{Label: "1", URL: "http://x/1.html"})
if err != nil {
t.Fatalf("ParseChapter: %v", err)
}
if strings.Contains(ch.Paragraphs[0], "vị") {
t.Fatal("word appeared without its CSS rule; fixture no longer proves anything")
}
if want := "Nghe trưởng tử noi."; ch.Paragraphs[0] != want {
t.Errorf("paragraph 0 = %q, want %q", ch.Paragraphs[0], want)
}
}
func TestParseChapterDropsScriptsAndSpacers(t *testing.T) {
ch, err := ParseChapter([]byte(chapterFixture), ChapterRef{Label: "1", URL: "http://x/1.html"})
if err != nil {
t.Fatalf("ParseChapter: %v", err)
}
for _, p := range ch.Paragraphs {
if strings.Contains(p, "ads()") {
t.Errorf("script text leaked into paragraph %q", p)
}
if strings.TrimSpace(p) == "" {
t.Error("empty spacer paragraph was kept")
}
if strings.Contains(p, " ") {
t.Errorf("non-breaking space survived in %q", p)
}
}
}
// The leading "<p>1</p>" repeats the chapter label and would print twice.
func TestParseChapterDropsRepeatedTitle(t *testing.T) {
ch, err := ParseChapter([]byte(chapterFixture), ChapterRef{Label: "1", URL: "http://x/1.html"})
if err != nil {
t.Fatalf("ParseChapter: %v", err)
}
if ch.Paragraphs[0] == "1" {
t.Error("repeated chapter label was kept as a paragraph")
}
}
// gatedChapterFixture mirrors how most chapters are served: a visible sponsor
// block (div.actcl) stands in for the body, while the real text sits in a
// sibling div.actac hidden with display:none and revealed by the site's
// JavaScript. A source watermark and prev/next navigation bracket the prose.
const gatedChapterFixture = `<!DOCTYPE html><html><head>
<style>.t-aaa:before { content: "v\1ecb"; }</style></head><body>
<div class="content-container" id="chapter-content-render">
<div class="actcl">
<h4 class="text-center text-primary">Moi Quy doc gia CLICK vao lien ket</h4>
<p class="text-center">mo ung dung Shopee, sau do quay tro lai de doc!</p>
<a class="btn btn-primary px-3" href="https://s.shopee.vn/xxxx">
<img src="x.jpg"><span class="text-uppercase text-danger">CLICK</span></a>
<h4 class="text-center text-primary">MonkeyD va doi ngu Editor xin chan thanh cam on!</h4>
</div>
<div class="actac" style=" display:none; ">
<p>Doan van dau tien co <span class="t-aaa"></span> tri.</p>
<p>&nbsp;</p>
<p class="signature">[Truyen duoc dang tai duy nhat tai MonkeyDD.com - https://monkeydd.com/n/10.html.]</p>
<p>Doan van cuoi cung.</p>
</div>
<div class="my-4"><div class="d-flex justify-content-center">
<a class="btn btn-primary px-3 me-2" href="/n/9.html"><i class="bx bx-chevron-left"></i>Chương trước</a>
<a class="btn btn-primary px-3" href="/n/11.html">Chương sau<i class="bx bx-chevron-right"></i></a>
</div></div>
</div></body></html>`
// The gated body is the one thing that must survive: it is hidden with
// display:none, so any rule that drops hidden or "act*" containers silently
// discards the entire chapter.
func TestParseChapterKeepsGatedBody(t *testing.T) {
ch, err := ParseChapter([]byte(gatedChapterFixture), ChapterRef{Label: "10", URL: "http://x/10.html"})
if err != nil {
t.Fatalf("ParseChapter: %v", err)
}
want := []string{
"Doan van dau tien co vị tri.",
"Doan van cuoi cung.",
}
if len(ch.Paragraphs) != len(want) {
t.Fatalf("got %d paragraphs %q, want %d", len(ch.Paragraphs), ch.Paragraphs, len(want))
}
for i, w := range want {
if ch.Paragraphs[i] != w {
t.Errorf("paragraph %d = %q, want %q", i, ch.Paragraphs[i], w)
}
}
}
// Everything the site injects around the prose must be gone.
func TestParseChapterDropsInjectedBlocks(t *testing.T) {
ch, err := ParseChapter([]byte(gatedChapterFixture), ChapterRef{Label: "10", URL: "http://x/10.html"})
if err != nil {
t.Fatalf("ParseChapter: %v", err)
}
body := strings.Join(ch.Paragraphs, "\n")
for _, junk := range []string{
"Shopee", // sponsor copy
"CLICK", // sponsor call to action
"cam on", // sponsor sign-off
"MonkeyDD.com", // source watermark
"Chương trước", // navigation
"Chương sau", // navigation
} {
if strings.Contains(body, junk) {
t.Errorf("injected text %q survived extraction in %q", junk, body)
}
}
}
func TestParseChapterMissingContainer(t *testing.T) {
if _, err := ParseChapter([]byte(`<html><body><p>hi</p></body></html>`),
ChapterRef{URL: "http://x/1.html"}); err == nil {
t.Fatal("want an error when the content container is absent")
}
}
func TestChapterHeading(t *testing.T) {
for _, tc := range []struct{ label, want string }{
{"14", "Chương 14"},
{"Chương 12", "Chương 12"},
{"", "Chương"},
} {
ch := &Chapter{Label: tc.label}
if got := ch.Heading(); got != tc.want {
t.Errorf("Heading(%q) = %q, want %q", tc.label, got, tc.want)
}
}
}
+135
View File
@@ -0,0 +1,135 @@
package crawler
import (
"context"
"errors"
"fmt"
"io"
"net/http"
"sync"
"time"
)
// The site rejects requests without a browser-like User-Agent.
const defaultUserAgent = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 " +
"(KHTML, like Gecko) Chrome/124.0.0.0 Safari/537.36"
// maxPageSize caps how much of a response we buffer; chapter pages are ~130 KB.
const maxPageSize = 8 << 20
// statusError reports an unexpected HTTP status so Get can decide whether
// retrying is worthwhile.
type statusError struct {
code int
status string
}
func (e *statusError) Error() string { return "unexpected status " + e.status }
// retryable is true for transient failures. A 404 or 403 will not fix itself,
// so those fail immediately instead of burning the retry budget.
func (e *statusError) retryable() bool {
return e.code == http.StatusTooManyRequests || e.code >= 500
}
// Client fetches pages from monkeydd.com. It spaces requests out by a fixed
// delay no matter how many goroutines call Get, so raising the worker count
// never raises the request rate, and it retries transient failures with
// exponential backoff.
type Client struct {
http *http.Client
ua string
delay time.Duration
retries int
mu sync.Mutex
nextSlot time.Time
}
func NewClient(delay time.Duration, retries int) *Client {
return &Client{
http: &http.Client{Timeout: 45 * time.Second},
ua: defaultUserAgent,
delay: delay,
retries: retries,
}
}
// reserve claims the next request slot and blocks until it comes due, holding
// the global rate at one request per delay across all callers.
func (c *Client) reserve(ctx context.Context) error {
c.mu.Lock()
slot := c.nextSlot
if now := time.Now(); slot.Before(now) {
slot = now
}
c.nextSlot = slot.Add(c.delay)
c.mu.Unlock()
wait := time.Until(slot)
if wait <= 0 {
return nil
}
timer := time.NewTimer(wait)
defer timer.Stop()
select {
case <-ctx.Done():
return ctx.Err()
case <-timer.C:
return nil
}
}
// Get fetches url, retrying transient failures.
func (c *Client) Get(ctx context.Context, url string) ([]byte, error) {
var lastErr error
for attempt := 0; attempt <= c.retries; attempt++ {
if attempt > 0 {
backoff := time.Duration(1<<uint(attempt-1)) * time.Second
select {
case <-ctx.Done():
return nil, ctx.Err()
case <-time.After(backoff):
}
}
if err := c.reserve(ctx); err != nil {
return nil, err
}
body, err := c.fetch(ctx, url)
if err == nil {
return body, nil
}
lastErr = err
var se *statusError
if errors.As(err, &se) && !se.retryable() {
break
}
if ctx.Err() != nil {
break
}
}
return nil, fmt.Errorf("fetch %s: %w", url, lastErr)
}
func (c *Client) fetch(ctx context.Context, url string) ([]byte, error) {
req, err := http.NewRequestWithContext(ctx, http.MethodGet, url, nil)
if err != nil {
return nil, err
}
req.Header.Set("User-Agent", c.ua)
req.Header.Set("Accept", "text/html,application/xhtml+xml,*/*;q=0.8")
req.Header.Set("Accept-Language", "vi,en;q=0.8")
resp, err := c.http.Do(req)
if err != nil {
return nil, err
}
defer func() { _ = resp.Body.Close() }()
if resp.StatusCode != http.StatusOK {
return nil, &statusError{code: resp.StatusCode, status: resp.Status}
}
return io.ReadAll(io.LimitReader(resp.Body, maxPageSize))
}
+198
View File
@@ -0,0 +1,198 @@
package crawler
import (
"context"
"fmt"
"net/url"
"os"
"path/filepath"
"regexp"
"strings"
"sync"
"golang.org/x/sync/errgroup"
)
// Crawler fetches a novel and its chapters.
type Crawler struct {
Client *Client
// CacheDir, when set, stores raw pages on disk and serves later runs from
// them. Re-exporting with different font or page settings then costs no
// requests.
CacheDir string
// Workers bounds concurrent fetches. The client's delay still caps the
// overall request rate.
Workers int
// Log receives progress messages. Optional.
Log func(format string, args ...any)
}
func (c *Crawler) logf(format string, args ...any) {
if c.Log != nil {
c.Log(format, args...)
}
}
// Novel fetches a novel landing page and resolves its chapter list.
//
// The chapter list is taken from the landing page and cross-checked against the
// dropdown embedded in the first chapter page, so a truncated list cannot
// silently shorten the export.
func (c *Crawler) Novel(ctx context.Context, novelURL string) (*Novel, error) {
page, err := c.page(ctx, novelURL)
if err != nil {
return nil, err
}
novel, err := ParseNovelPage(page, novelURL)
if err != nil {
return nil, err
}
c.logf("novel: %s (%d chapters listed)", novel.Title, len(novel.Chapters))
base, err := url.Parse(novelURL)
if err != nil {
return nil, err
}
firstPage, err := c.page(ctx, novel.Chapters[0].URL)
if err != nil {
return nil, err
}
fromSelect, err := ChapterRefsFromSelect(firstPage, base)
if err != nil {
return nil, err
}
final, extra := ReconcileChapterRefs(novel.Chapters, fromSelect)
if len(extra) > 0 {
c.logf("warning: %d chapter(s) appear only in the chapter dropdown and were not "+
"in the landing page list; verify the export is complete", len(extra))
}
if len(final) != len(novel.Chapters) {
c.logf("chapter list reconciled to %d chapters using the in-chapter dropdown", len(final))
}
novel.Chapters = final
return novel, nil
}
// NovelInfo fetches only the landing page and returns what it carries: title,
// slug, tags, and the chapter list as that page shows it.
//
// Unlike Novel it does not also fetch a chapter page to cross-check the chapter
// list, so it costs a single request. Callers that only want metadata should
// prefer it; callers about to export every chapter want Novel, whose
// reconciliation guards against a silently truncated list.
func (c *Crawler) NovelInfo(ctx context.Context, novelURL string) (*Novel, error) {
page, err := c.page(ctx, novelURL)
if err != nil {
return nil, err
}
return ParseNovelPage(page, novelURL)
}
// Chapters fetches every chapter concurrently and returns them in reading
// order. Any chapter that cannot be fetched or parsed fails the whole run
// rather than yielding a book with a hole in it.
func (c *Crawler) Chapters(ctx context.Context, novel *Novel) ([]*Chapter, error) {
chapters := make([]*Chapter, len(novel.Chapters))
workers := c.Workers
if workers < 1 {
workers = 1
}
group, groupCtx := errgroup.WithContext(ctx)
group.SetLimit(workers)
var mu sync.Mutex
done := 0
for i, ref := range novel.Chapters {
i, ref := i, ref
group.Go(func() error {
page, err := c.page(groupCtx, ref.URL)
if err != nil {
return err
}
chapter, err := ParseChapter(page, ref)
if err != nil {
return err
}
chapters[i] = chapter
mu.Lock()
done++
c.logf("fetched %d/%d: %s (%d words)", done, len(novel.Chapters),
chapter.Heading(), chapter.WordCount())
mu.Unlock()
return nil
})
}
if err := group.Wait(); err != nil {
return nil, err
}
return chapters, nil
}
// page returns a page from the cache when available, otherwise fetches and
// caches it.
func (c *Crawler) page(ctx context.Context, pageURL string) ([]byte, error) {
path := c.cachePath(pageURL)
if path != "" {
if body, err := os.ReadFile(path); err == nil && len(body) > 0 { //nolint:gosec // G304: cachePath strips every path separator from the URL
return body, nil
}
}
body, err := c.Client.Get(ctx, pageURL)
if err != nil {
return nil, err
}
if path != "" {
if err := os.MkdirAll(filepath.Dir(path), 0o750); err == nil {
// A failed cache write must not fail the crawl.
_ = os.WriteFile(path, body, 0o600)
}
}
return body, nil
}
// unsafeFileChars matches everything not allowed in a cache file name.
var unsafeFileChars = regexp.MustCompile(`[^A-Za-z0-9._-]+`)
// cachePath maps a page URL to a cache file, or "" when caching is disabled.
func (c *Crawler) cachePath(pageURL string) string {
if c.CacheDir == "" {
return ""
}
u, err := url.Parse(pageURL)
if err != nil {
return ""
}
name := unsafeFileChars.ReplaceAllString(strings.Trim(u.Path, "/"), "_")
if name == "" {
return ""
}
if !strings.HasSuffix(name, ".html") {
name += ".html"
}
return filepath.Join(c.CacheDir, name)
}
// TotalWords sums the word count across chapters.
func TotalWords(chapters []*Chapter) int {
n := 0
for _, ch := range chapters {
n += ch.WordCount()
}
return n
}
// Describe renders a one-line summary of a crawl result.
func Describe(novel *Novel, chapters []*Chapter) string {
return fmt.Sprintf("%s — %d chapters, %d words", novel.Title, len(chapters), TotalWords(chapters))
}
@@ -0,0 +1,50 @@
package crawler
import (
"regexp"
"strconv"
"strings"
)
// The site hides part of every chapter behind CSS rather than putting it in the
// markup. Chapter HTML carries empty elements such as
//
// Nghe <span class="t-3e625e..."></span> trưởng tử
//
// and the page stylesheet supplies the missing word:
//
// .t-3e625e...:before { content: "vị"; }
//
// Reading DOM text alone therefore drops hundreds of words per chapter without
// any visible error. wordRule finds those rules so the words can be put back.
var wordRule = regexp.MustCompile(`\.([A-Za-z0-9_-]+)\s*::?before\s*\{[^}]*?content\s*:\s*"((?:[^"\\]|\\.)*)"`)
// cssEscape matches a CSS character escape: a hex code point, optionally
// followed by one whitespace terminator, or an escaped literal character.
var cssEscape = regexp.MustCompile(`\\([0-9A-Fa-f]{1,6})\s?|\\(.)`)
// ParseWordClasses maps CSS class name to the word its :before rule injects.
func ParseWordClasses(page []byte) map[string]string {
words := make(map[string]string)
for _, m := range wordRule.FindAllSubmatch(page, -1) {
words[string(m[1])] = decodeCSSString(string(m[2]))
}
return words
}
// decodeCSSString resolves the escape sequences allowed inside a CSS string.
func decodeCSSString(s string) string {
if !strings.Contains(s, `\`) {
return s
}
return cssEscape.ReplaceAllStringFunc(s, func(esc string) string {
m := cssEscape.FindStringSubmatch(esc)
if m[1] != "" {
if cp, err := strconv.ParseInt(m[1], 16, 32); err == nil && cp > 0 {
return string(rune(cp))
}
return ""
}
return m[2]
})
}
@@ -0,0 +1,95 @@
package crawler
import (
"strings"
"golang.org/x/net/html"
)
// attr returns the value of the named attribute, or "" when absent.
func attr(n *html.Node, name string) string {
for _, a := range n.Attr {
if a.Key == name {
return a.Val
}
}
return ""
}
// hasClass reports whether the node carries the given class token.
func hasClass(n *html.Node, class string) bool {
for _, tok := range strings.Fields(attr(n, "class")) {
if tok == class {
return true
}
}
return false
}
// findNode returns the first node in document order satisfying match.
func findNode(root *html.Node, match func(*html.Node) bool) *html.Node {
if match(root) {
return root
}
for c := root.FirstChild; c != nil; c = c.NextSibling {
if found := findNode(c, match); found != nil {
return found
}
}
return nil
}
// findAllNodes returns every node satisfying match, in document order.
func findAllNodes(root *html.Node, match func(*html.Node) bool) []*html.Node {
var out []*html.Node
var walk func(*html.Node)
walk = func(n *html.Node) {
if match(n) {
out = append(out, n)
}
for c := n.FirstChild; c != nil; c = c.NextSibling {
walk(c)
}
}
walk(root)
return out
}
// elementByID finds an element by its id attribute.
func elementByID(root *html.Node, id string) *html.Node {
return findNode(root, func(n *html.Node) bool {
return n.Type == html.ElementNode && attr(n, "id") == id
})
}
// elementByTag finds the first element with the given tag name.
func elementByTag(root *html.Node, tag string) *html.Node {
return findNode(root, func(n *html.Node) bool {
return n.Type == html.ElementNode && n.Data == tag
})
}
// nodeText collects the descendant text of a node with whitespace collapsed.
func nodeText(n *html.Node) string {
if n == nil {
return ""
}
var b strings.Builder
var walk func(*html.Node)
walk = func(n *html.Node) {
if n.Type == html.TextNode {
b.WriteString(n.Data)
}
for c := n.FirstChild; c != nil; c = c.NextSibling {
walk(c)
}
}
walk(n)
return collapseSpaces(b.String())
}
// collapseSpaces trims the string and reduces every whitespace run, including
// the non-breaking spaces the site emits as &nbsp;, to a single space.
func collapseSpaces(s string) string {
return strings.Join(strings.Fields(s), " ")
}
+223
View File
@@ -0,0 +1,223 @@
package crawler
import (
"bytes"
"fmt"
"net/url"
"strings"
"golang.org/x/net/html"
)
// ChapterRef points at a single chapter listed on a novel page.
type ChapterRef struct {
Label string // as shown on the site, e.g. "14" or "Chương 12"
URL string
}
// Novel is a novel's landing page: its title and its chapters in reading order.
type Novel struct {
Title string
Slug string
URL string
// Tags are the novel's own genres, as labelled on the site, in the order
// they appear. Empty when the page lists none.
Tags []string
Chapters []ChapterRef
}
// ParseNovelPage reads the title and chapter list from a novel landing page.
//
// Chapter URLs are always taken from the anchors on the page. Slugs are not
// uniform across novels ("/14.html" on one, "/chuong-12.html" on another) and
// numbering has gaps, so generating URLs from a chapter count would fetch 404s
// and miss real chapters.
func ParseNovelPage(page []byte, pageURL string) (*Novel, error) {
doc, err := html.Parse(bytes.NewReader(page))
if err != nil {
return nil, fmt.Errorf("parse novel page: %w", err)
}
base, err := url.Parse(pageURL)
if err != nil {
return nil, fmt.Errorf("parse novel url: %w", err)
}
novel := &Novel{
Title: novelTitle(doc),
Slug: slugFromNovelURL(base),
URL: pageURL,
Tags: tagsFromInfo(doc),
Chapters: chapterRefsFromList(doc, base),
}
if novel.Title == "" {
novel.Title = novel.Slug
}
if len(novel.Chapters) == 0 {
return nil, fmt.Errorf("no chapters found on %s (page layout may have changed)", pageURL)
}
return novel, nil
}
// novelTitle prefers the <h1> heading and falls back to the document title.
func novelTitle(doc *html.Node) string {
if h1 := elementByTag(doc, "h1"); h1 != nil {
if t := nodeText(h1); t != "" {
return t
}
}
return nodeText(elementByTag(doc, "title"))
}
// tagsFromInfo reads the novel's own genres out of the info block.
//
// The anchors are matched on the schema.org itemprop="genre" microdata rather
// than on their href. The page also carries a site-wide genre menu linking every
// category — 69 distinct ones against this novel's 6 on a sampled page — so
// matching the /the-loai/ URL shape would pull in the whole navigation.
func tagsFromInfo(doc *html.Node) []string {
links := findAllNodes(doc, func(n *html.Node) bool {
return n.Type == html.ElementNode && n.Data == "a" && attr(n, "itemprop") == "genre"
})
var tags []string
seen := make(map[string]bool, len(links))
for _, link := range links {
// The label appears both as the link text and as its title attribute;
// prefer the text and fall back to the attribute.
name := nodeText(link)
if name == "" {
name = collapseSpaces(attr(link, "title"))
}
if name == "" || seen[name] {
continue
}
seen[name] = true
tags = append(tags, name)
}
return tags
}
// slugFromNovelURL turns https://host/tro-lai-nam-thang-cu.html into
// "tro-lai-nam-thang-cu".
func slugFromNovelURL(u *url.URL) string {
seg := strings.Trim(u.Path, "/")
if i := strings.LastIndex(seg, "/"); i >= 0 {
seg = seg[i+1:]
}
return strings.TrimSuffix(seg, ".html")
}
// chapterRefsFromList reads the "list-chapters" block on the landing page.
// The site lists newest first, so the result is reversed into reading order.
func chapterRefsFromList(doc *html.Node, base *url.URL) []ChapterRef {
list := findNode(doc, func(n *html.Node) bool {
return n.Type == html.ElementNode && hasClass(n, "list-chapters")
})
if list == nil {
return nil
}
titles := findAllNodes(list, func(n *html.Node) bool {
return n.Type == html.ElementNode && hasClass(n, "episode-title")
})
var refs []ChapterRef
for _, title := range titles {
link := elementByTag(title, "a")
if link == nil {
continue
}
href := strings.TrimSpace(attr(link, "href"))
if href == "" {
continue
}
abs, err := base.Parse(href)
if err != nil {
continue
}
refs = append(refs, ChapterRef{Label: nodeText(link), URL: abs.String()})
}
return reverseRefs(refs)
}
// ChapterRefsFromSelect reads the chapter dropdown embedded in every chapter
// page, whose options hold "novel-slug,chapter-slug" pairs. This is a second,
// independent view of the chapter list used to cross-check the landing page.
func ChapterRefsFromSelect(page []byte, base *url.URL) ([]ChapterRef, error) {
doc, err := html.Parse(bytes.NewReader(page))
if err != nil {
return nil, fmt.Errorf("parse chapter page: %w", err)
}
sel := elementByID(doc, "selected_chapter")
if sel == nil {
return nil, nil
}
var refs []ChapterRef
for _, opt := range findAllNodes(sel, func(n *html.Node) bool {
return n.Type == html.ElementNode && n.Data == "option"
}) {
novelSlug, chapterSlug, ok := strings.Cut(attr(opt, "value"), ",")
if !ok || novelSlug == "" || chapterSlug == "" {
continue
}
abs, err := base.Parse("/" + novelSlug + "/" + chapterSlug + ".html")
if err != nil {
continue
}
refs = append(refs, ChapterRef{Label: nodeText(opt), URL: abs.String()})
}
return reverseRefs(refs), nil
}
// ReconcileChapterRefs picks the chapter list to crawl from the landing page
// list and the in-chapter dropdown.
//
// Both views come from the same site ordering, so when the dropdown is a
// superset it is preferred: that keeps the run correct even if the landing page
// ever truncates or paginates its list. Anything the dropdown alone knows about
// while disagreeing on order is returned as extra so the caller can warn rather
// than silently export a short book.
func ReconcileChapterRefs(fromList, fromSelect []ChapterRef) (final, extra []ChapterRef) {
if len(fromSelect) == 0 {
return fromList, nil
}
inSelect := refURLSet(fromSelect)
if listIsSubsetOf(fromList, inSelect) {
return fromSelect, nil
}
inList := refURLSet(fromList)
for _, ref := range fromSelect {
if !inList[ref.URL] {
extra = append(extra, ref)
}
}
return fromList, extra
}
func refURLSet(refs []ChapterRef) map[string]bool {
set := make(map[string]bool, len(refs))
for _, r := range refs {
set[r.URL] = true
}
return set
}
func listIsSubsetOf(refs []ChapterRef, set map[string]bool) bool {
for _, r := range refs {
if !set[r.URL] {
return false
}
}
return true
}
func reverseRefs(refs []ChapterRef) []ChapterRef {
for i, j := 0, len(refs)-1; i < j; i, j = i+1, j-1 {
refs[i], refs[j] = refs[j], refs[i]
}
return refs
}
@@ -0,0 +1,170 @@
package crawler
import (
"net/url"
"testing"
)
// novelFixture reproduces the two traits that break naive crawlers: the list is
// newest-first, and chapter numbering has a gap (no chapter 4).
const novelFixture = `<html><head><title>TEN TRUYEN</title></head><body>
<h1>TEN TRUYEN</h1>
<div class="list-chapters">
<div class="item"><div class="episode-title"><a href="https://monkeydd.com/n/5.html">5</a></div></div>
<div class="item"><div class="episode-title"><a href="https://monkeydd.com/n/3.html">3</a></div></div>
<div class="item"><div class="episode-title"><a href="/n/2.html">2</a></div></div>
<div class="item"><div class="episode-title"><a href="https://monkeydd.com/n/1.html">1</a></div></div>
</div></body></html>`
func TestParseNovelPageOrdersChaptersForReading(t *testing.T) {
novel, err := ParseNovelPage([]byte(novelFixture), "https://monkeydd.com/n.html")
if err != nil {
t.Fatalf("ParseNovelPage: %v", err)
}
if novel.Title != "TEN TRUYEN" {
t.Errorf("Title = %q", novel.Title)
}
if novel.Slug != "n" {
t.Errorf("Slug = %q, want %q", novel.Slug, "n")
}
wantLabels := []string{"1", "2", "3", "5"}
if len(novel.Chapters) != len(wantLabels) {
t.Fatalf("got %d chapters, want %d", len(novel.Chapters), len(wantLabels))
}
for i, want := range wantLabels {
if novel.Chapters[i].Label != want {
t.Errorf("chapter %d label = %q, want %q", i, novel.Chapters[i].Label, want)
}
}
// Relative hrefs must resolve against the novel URL.
if got, want := novel.Chapters[1].URL, "https://monkeydd.com/n/2.html"; got != want {
t.Errorf("chapter 2 URL = %q, want %q", got, want)
}
}
// tagsFixture mirrors the real page: the novel's own genres carry
// itemprop="genre", while a site-wide menu links every category without it. A
// parser matching on the /the-loai/ href shape would swallow the whole menu.
const tagsFixture = `<html><head><title>TEN TRUYEN</title></head><body>
<h1>TEN TRUYEN</h1>
<dl class="row">
<dt class="col-sm-3">Thể loại</dt>
<dd class="col-sm-9">
<a class='cate-item' itemprop='genre' title='Trọng Sinh' href='https://monkeydd.com/the-loai/trong-sinh.html'>Trọng Sinh</a>
<a class='cate-item' itemprop='genre' title='Cổ Đại' href='https://monkeydd.com/the-loai/co-dai.html'>Cổ Đại</a>
<a class='cate-item' itemprop='genre' title='Gia Đình' href='https://monkeydd.com/the-loai/gia-dinh.html'></a>
<a class='cate-item' itemprop='genre' title='Cổ Đại' href='https://monkeydd.com/the-loai/co-dai.html'>Cổ Đại</a>
</dd>
</dl>
<nav class="site-menu">
<a href='https://monkeydd.com/the-loai/dam-my.html'>Đam Mỹ</a>
<a href='https://monkeydd.com/the-loai/bach-hop.html'>Bách Hợp</a>
</nav>
<div class="list-chapters">
<div class="item"><div class="episode-title"><a href="/n/1.html">1</a></div></div>
</div></body></html>`
func TestParseNovelPageReadsTags(t *testing.T) {
novel, err := ParseNovelPage([]byte(tagsFixture), "https://monkeydd.com/n.html")
if err != nil {
t.Fatalf("ParseNovelPage: %v", err)
}
// "Cổ Đại" has its inner whitespace collapsed, the empty anchor falls back
// to its title attribute, and the repeated genre appears once.
want := []string{"Trọng Sinh", "Cổ Đại", "Gia Đình"}
if len(novel.Tags) != len(want) {
t.Fatalf("Tags = %q, want %q", novel.Tags, want)
}
for i := range want {
if novel.Tags[i] != want[i] {
t.Errorf("Tags[%d] = %q, want %q", i, novel.Tags[i], want[i])
}
}
}
func TestParseNovelPageWithoutTags(t *testing.T) {
novel, err := ParseNovelPage([]byte(novelFixture), "https://monkeydd.com/n.html")
if err != nil {
t.Fatalf("ParseNovelPage: %v", err)
}
if len(novel.Tags) != 0 {
t.Errorf("Tags = %q, want none", novel.Tags)
}
}
func TestParseNovelPageNoChapters(t *testing.T) {
if _, err := ParseNovelPage([]byte(`<html><title>x</title><body></body></html>`),
"https://monkeydd.com/n.html"); err == nil {
t.Fatal("want an error when no chapters are listed")
}
}
const selectFixture = `<html><body>
<select name="selected_chapter" id="selected_chapter">
<option value="n,5">5</option>
<option value="n,chuong-3">Chương 3</option>
<option value="n,1">1</option>
<option value="14">malformed</option>
</select></body></html>`
func TestChapterRefsFromSelect(t *testing.T) {
base, err := url.Parse("https://monkeydd.com/n.html")
if err != nil {
t.Fatal(err)
}
refs, err := ChapterRefsFromSelect([]byte(selectFixture), base)
if err != nil {
t.Fatalf("ChapterRefsFromSelect: %v", err)
}
if len(refs) != 3 {
t.Fatalf("got %d refs %+v, want 3 (malformed option skipped)", len(refs), refs)
}
if got, want := refs[0].URL, "https://monkeydd.com/n/1.html"; got != want {
t.Errorf("first ref URL = %q, want %q", got, want)
}
if got, want := refs[1].URL, "https://monkeydd.com/n/chuong-3.html"; got != want {
t.Errorf("second ref URL = %q, want %q", got, want)
}
}
func TestReconcileChapterRefs(t *testing.T) {
ref := func(u string) ChapterRef { return ChapterRef{Label: u, URL: u} }
t.Run("dropdown superset wins", func(t *testing.T) {
list := []ChapterRef{ref("a"), ref("b")}
sel := []ChapterRef{ref("a"), ref("b"), ref("c")}
final, extra := ReconcileChapterRefs(list, sel)
if len(final) != 3 {
t.Errorf("got %d chapters, want the 3 from the dropdown", len(final))
}
if len(extra) != 0 {
t.Errorf("got %d extra, want 0", len(extra))
}
})
t.Run("disagreement is reported not silently merged", func(t *testing.T) {
list := []ChapterRef{ref("a"), ref("z")}
sel := []ChapterRef{ref("a"), ref("b")}
final, extra := ReconcileChapterRefs(list, sel)
if len(final) != 2 || final[1].URL != "z" {
t.Errorf("final = %+v, want the landing page list", final)
}
if len(extra) != 1 || extra[0].URL != "b" {
t.Errorf("extra = %+v, want [b] so the caller can warn", extra)
}
})
t.Run("empty dropdown falls back to the list", func(t *testing.T) {
list := []ChapterRef{ref("a")}
final, extra := ReconcileChapterRefs(list, nil)
if len(final) != 1 || len(extra) != 0 {
t.Errorf("final = %+v, extra = %+v", final, extra)
}
})
}
+154
View File
@@ -0,0 +1,154 @@
// Package export turns a novel URL into a PDF file: resolve the chapter list,
// fetch the chapters, render the PDF with the bundled font.
package export
import (
"context"
"fmt"
"net/url"
"path/filepath"
"regexp"
"strings"
"time"
"github.com/tiennm99/miti99bot/internal/modules/monkeyd/crawler"
"github.com/tiennm99/miti99bot/internal/modules/monkeyd/pdf"
)
// Defaults the bot advertises to users or reuses in other crawls.
const (
DefaultFontSize = 10.0
DefaultDelay = 400 * time.Millisecond
)
// Fixed layout and crawl settings. The bot exposes only the font size.
const (
lineSpacing = 1.55 // multiple of font size
margin = 6.0 // millimetres
workers = 4
retries = 3
)
// Request describes one export.
type Request struct {
NovelURL string
// OutDir receives the PDF, named after the novel title.
OutDir string
// FontSize is in points. 0 means DefaultFontSize.
FontSize float64
// CacheDir stores raw pages so a re-export costs no requests. Empty
// disables the cache.
CacheDir string
// Log receives progress messages. Optional.
Log func(format string, args ...any)
}
// Result reports what was produced.
type Result struct {
Path string
Title string
SourceURL string
Chapters int
Words int
Page pdf.PageSize
}
// Summary renders a one-line description of the exported book.
func (r *Result) Summary() string {
return fmt.Sprintf("%s — %d chapters, %d words", r.Title, r.Chapters, r.Words)
}
// Export fetches the novel at req.NovelURL and writes it as a PDF, returning
// where it landed. The context bounds the whole crawl; cancelling it abandons
// the run without leaving a partial PDF behind.
func Export(ctx context.Context, req Request) (*Result, error) {
if req.FontSize == 0 {
req.FontSize = DefaultFontSize
}
if err := req.validate(); err != nil {
return nil, err
}
c := &crawler.Crawler{
Client: crawler.NewClient(DefaultDelay, retries),
CacheDir: req.CacheDir,
Workers: workers,
Log: req.Log,
}
novel, err := c.Novel(ctx, req.NovelURL)
if err != nil {
return nil, err
}
chapters, err := c.Chapters(ctx, novel)
if err != nil {
return nil, err
}
outPath := filepath.Join(req.OutDir, SafeFileName(novel.Title, novel.Slug)+".pdf")
opts := pdf.Options{
Page: pdf.PhonePage,
Margin: margin,
Font: pdf.BundledFont(),
FontSize: req.FontSize,
LineSpacing: lineSpacing,
Title: novel.Title,
SourceURL: novel.URL,
}
if err := pdf.Write(outPath, opts, toPDFChapters(chapters)); err != nil {
return nil, err
}
return &Result{
Path: outPath,
Title: novel.Title,
SourceURL: novel.URL,
Chapters: len(chapters),
Words: crawler.TotalWords(chapters),
Page: pdf.PhonePage,
}, nil
}
// validate rejects a Request before any request is made, so a typo costs no
// fetches.
func (r *Request) validate() error {
if r.NovelURL == "" {
return fmt.Errorf("novel url is required")
}
parsed, err := url.Parse(r.NovelURL)
if err != nil {
return fmt.Errorf("invalid novel url: %w", err)
}
if parsed.Scheme != "http" && parsed.Scheme != "https" {
return fmt.Errorf("invalid novel url: want an http(s) URL, got %q", r.NovelURL)
}
if r.FontSize <= 0 {
return fmt.Errorf("font size must be positive")
}
return nil
}
func toPDFChapters(chapters []*crawler.Chapter) []pdf.Chapter {
out := make([]pdf.Chapter, 0, len(chapters))
for _, ch := range chapters {
out = append(out, pdf.Chapter{Heading: ch.Heading(), Paragraphs: ch.Paragraphs})
}
return out
}
var unsafeNameChars = regexp.MustCompile(`[^\p{L}\p{N}]+`)
// SafeFileName builds a file name from the novel title, falling back to the
// slug when the title has no usable characters. The result has no extension.
func SafeFileName(title, fallback string) string {
name := strings.Trim(unsafeNameChars.ReplaceAllString(title, "-"), "-")
if name == "" {
return fallback
}
return name
}
@@ -0,0 +1,81 @@
package export
import "testing"
func TestValidate(t *testing.T) {
tests := []struct {
name string
req Request
wantErr bool
}{
{
name: "url and font size are valid",
req: Request{NovelURL: "https://monkeydd.com/example.html", FontSize: DefaultFontSize},
},
{
name: "missing url",
req: Request{FontSize: DefaultFontSize},
wantErr: true,
},
{
name: "non-http scheme",
req: Request{NovelURL: "ftp://monkeydd.com/example.html", FontSize: DefaultFontSize},
wantErr: true,
},
{
name: "scheme-less url",
req: Request{NovelURL: "monkeydd.com/example.html", FontSize: DefaultFontSize},
wantErr: true,
},
{
name: "negative font size",
req: Request{NovelURL: "https://monkeydd.com/example.html", FontSize: -1},
wantErr: true,
},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
err := tt.req.validate()
if tt.wantErr && err == nil {
t.Error("validate() = nil, want error")
}
if !tt.wantErr && err != nil {
t.Errorf("validate() = %v, want nil", err)
}
})
}
}
// An invalid request must fail before the crawler makes any request.
func TestExportRejectsInvalidRequestWithoutFetching(t *testing.T) {
if _, err := Export(t.Context(), Request{NovelURL: "ftp://monkeydd.com/x.html", OutDir: t.TempDir()}); err == nil {
t.Fatal("Export = nil error, want an invalid url error")
}
}
func TestSafeFileName(t *testing.T) {
tests := []struct {
title string
fallback string
want string
}{
{"Trở Lại Năm Tháng Cũ", "slug", "Trở-Lại-Năm-Tháng-Cũ"},
{"Chapter: One / Two", "slug", "Chapter-One-Two"},
{" --- ", "slug", "slug"},
{"", "slug", "slug"},
}
for _, tt := range tests {
if got := SafeFileName(tt.title, tt.fallback); got != tt.want {
t.Errorf("SafeFileName(%q, %q) = %q, want %q", tt.title, tt.fallback, got, tt.want)
}
}
}
func TestResultSummary(t *testing.T) {
r := &Result{Title: "Example", Chapters: 12, Words: 3400}
want := "Example — 12 chapters, 3400 words"
if got := r.Summary(); got != want {
t.Errorf("Summary() = %q, want %q", got, want)
}
}
+1 -1
View File
@@ -11,7 +11,7 @@ import (
"github.com/go-telegram/bot"
"github.com/go-telegram/bot/models"
"github.com/tiennm99/monkeyd-crawler/export"
"github.com/tiennm99/miti99bot/internal/modules/monkeyd/export"
"github.com/tiennm99/miti99bot/internal/log"
"github.com/tiennm99/miti99bot/internal/modules/util/chathelper"
+3 -3
View File
@@ -9,8 +9,8 @@ import (
"strings"
"testing"
"github.com/tiennm99/monkeyd-crawler/export"
"github.com/tiennm99/monkeyd-crawler/pdfout"
"github.com/tiennm99/miti99bot/internal/modules/monkeyd/export"
"github.com/tiennm99/miti99bot/internal/modules/monkeyd/pdf"
"github.com/tiennm99/miti99bot/internal/modules"
"github.com/tiennm99/miti99bot/internal/storage"
@@ -77,7 +77,7 @@ func stubPDF(t *testing.T, dir, name string, size int64) *export.Result {
SourceURL: testNovelURL,
Chapters: 3,
Words: 1200,
Page: pdfout.Presets["phone"],
Page: pdf.PhonePage,
}
}
+1 -1
View File
@@ -15,7 +15,7 @@ import (
"github.com/go-telegram/bot"
"github.com/go-telegram/bot/models"
"github.com/tiennm99/monkeyd-crawler/export"
"github.com/tiennm99/miti99bot/internal/modules/monkeyd/export"
"github.com/tiennm99/miti99bot/internal/modules"
"github.com/tiennm99/miti99bot/internal/modules/util/chathelper"
+37
View File
@@ -0,0 +1,37 @@
package pdf
import (
_ "embed"
)
// bundledTTF is the only font the bot renders with. A minimal container has no
// system fonts, and a single embedded font keeps the PDF identical on every
// host. See fonts/NOTICE.md for provenance and licensing.
//
// Vietnamese text needs the Latin Extended Additional block (ư, ạ, ế, ộ …);
// DejaVu Sans covers it, where a basic-Latin font would silently drop the
// diacritics.
//
//go:embed fonts/DejaVuSans.ttf
var bundledTTF []byte
// BundledFontName labels the embedded font in diagnostics. It is not a path;
// the font is compiled into the binary.
const BundledFontName = "DejaVu Sans (bundled)"
// Font is font data ready to embed in a PDF.
//
// The data is carried as bytes rather than as a path because fpdf joins a font
// path onto its own font directory, which it defaults to "." — turning an
// absolute path into a working-directory-relative one that resolves only when
// the process happens to run from the filesystem root.
type Font struct {
// Name identifies the font for diagnostics.
Name string
Data []byte
}
// BundledFont returns the font compiled into the binary.
func BundledFont() Font {
return Font{Name: BundledFontName, Data: bundledTTF}
}
+38
View File
@@ -0,0 +1,38 @@
package pdf
import (
"testing"
"golang.org/x/image/font/sfnt"
)
func TestBundledFontIsParseable(t *testing.T) {
font := BundledFont()
if font.Name != BundledFontName {
t.Errorf("Name = %q, want %q", font.Name, BundledFontName)
}
if _, err := sfnt.Parse(font.Data); err != nil {
t.Fatalf("bundled font does not parse as a TrueType font: %v", err)
}
}
// The bundled font exists to render Vietnamese. A replacement that lacks these
// glyphs would silently emit blanks, so check before trusting it.
func TestBundledFontCoversVietnamese(t *testing.T) {
parsed, err := sfnt.Parse(BundledFont().Data)
if err != nil {
t.Fatalf("parse bundled font: %v", err)
}
var buf sfnt.Buffer
for _, r := range []rune{'ư', 'ơ', 'đ', 'ạ', 'ế', 'ộ', 'ữ', 'ằ', 'ỷ', 'ỹ', 'Ọ', 'Ế', 'Ư', 'Đ'} {
index, err := parsed.GlyphIndex(&buf, r)
if err != nil {
t.Errorf("GlyphIndex(%q): %v", r, err)
continue
}
// Glyph 0 is .notdef — the character is absent from the font.
if index == 0 {
t.Errorf("bundled font has no glyph for %q (U+%04X)", r, r)
}
}
}
Binary file not shown.
@@ -0,0 +1,24 @@
# Bundled font
`DejaVuSans.ttf` is embedded into the binary and used when no font is supplied
and no suitable system font is found. It covers the Latin Extended Additional
block, which is what Vietnamese diacritics need — a font with only basic Latin
coverage silently drops them.
The following is recorded in the font file's own name table:
- Version: `Version 2.37`
- Copyright: `Copyright (c) 2003 by Bitstream, Inc. All Rights Reserved.`
`Copyright (c) 2006 by Tavmjong Bah. All Rights Reserved.`
`DejaVu changes are in public domain`
- License information: <http://dejavu.sourceforge.net/wiki/index.php/License>
The DejaVu fonts are free and redistributable, which is why they ship with most
Linux distributions. This copy came from Alpine's `font-dejavu` package, which
does not include the license text as a separate file. For a vendored copy of the
full license text, take `LICENSE` from the upstream DejaVu release and add it to
this directory.
To swap the bundled font, replace `DejaVuSans.ttf` and update this file. Verify
the replacement covers Vietnamese first — `TestBundledFontCoversVietnamese` in
`../font_test.go` checks a representative set of characters.
+132
View File
@@ -0,0 +1,132 @@
package pdf
import (
"fmt"
"strings"
"github.com/go-pdf/fpdf"
)
// pointsToMM converts typographic points to millimetres.
const pointsToMM = 25.4 / 72.0
// bodyFont is the internal family name registered with the PDF.
const bodyFont = "body"
// PageSize is a page in millimetres.
type PageSize struct {
Name string
W, H float64
}
// PhonePage is the page every export uses.
//
// Phone reading depends far more on page shape than on font size. A phone
// viewer scales a whole page to fit the screen, so a large font on an A4 page
// still ends up tiny: the page is ~3x wider than the screen and gets shrunk to
// match. A page cut to the phone's own aspect ratio fills the screen at 100%,
// which is why this is a small 9:16 page rather than A4 with big type.
var PhonePage = PageSize{Name: "phone", W: 90, H: 160}
// Chapter is a chapter ready to render.
type Chapter struct {
Heading string
Paragraphs []string
}
// Options controls the exported PDF.
type Options struct {
Page PageSize
Margin float64 // mm
Font Font // BundledFont()
FontSize float64 // pt
LineSpacing float64 // multiple of font size
Title string
SourceURL string
}
// footerReserve is the vertical space kept clear for the page number.
const footerReserve = 6.0
// Write renders the chapters to a PDF at path.
func Write(path string, opts Options, chapters []Chapter) error {
pdf := fpdf.NewCustom(&fpdf.InitType{
UnitStr: "mm",
Size: fpdf.SizeType{Wd: opts.Page.W, Ht: opts.Page.H},
})
pdf.SetMargins(opts.Margin, opts.Margin, opts.Margin)
pdf.SetAutoPageBreak(true, opts.Margin+footerReserve)
// Embeds a subset of the TrueType data, which is what makes the Vietnamese
// diacritics render instead of falling back to "?". The bytes are passed
// directly rather than by path: the path-taking variant joins the name onto
// fpdf's own font directory (default "."), which mangles an absolute path
// into a working-directory-relative one.
pdf.AddUTF8FontFromBytes(bodyFont, "", opts.Font.Data)
pdf.SetFont(bodyFont, "", opts.FontSize)
pdf.SetTitle(opts.Title, true)
lineHeight := opts.FontSize * opts.LineSpacing * pointsToMM
paragraphGap := lineHeight * 0.45
headingSize := opts.FontSize * 1.35
addFooter(pdf, opts)
writeTitlePage(pdf, opts, len(chapters))
for _, ch := range chapters {
pdf.AddPage()
pdf.SetFontSize(headingSize)
pdf.MultiCell(0, headingSize*1.3*pointsToMM, ch.Heading, "", "L", false)
pdf.Ln(paragraphGap * 1.6)
pdf.SetFontSize(opts.FontSize)
for _, p := range ch.Paragraphs {
// "J" justifies, which keeps the short measure of a phone page tidy.
pdf.MultiCell(0, lineHeight, p, "", "J", false)
pdf.Ln(paragraphGap)
}
}
if err := pdf.OutputFileAndClose(path); err != nil {
return fmt.Errorf("write pdf %s: %w", path, err)
}
return nil
}
// addFooter prints a centred page number, restoring the body font size so the
// footer callback cannot leak its own size into the following content.
func addFooter(pdf *fpdf.Fpdf, opts Options) {
pdf.SetFooterFunc(func() {
if pdf.PageNo() <= 1 {
return
}
pdf.SetY(-(opts.Margin + footerReserve*0.6))
pdf.SetFontSize(opts.FontSize * 0.75)
pdf.SetTextColor(120, 120, 120)
pdf.CellFormat(0, 4, fmt.Sprintf("%d", pdf.PageNo()-1), "", 0, "C", false, 0, "")
pdf.SetTextColor(0, 0, 0)
pdf.SetFontSize(opts.FontSize)
})
}
func writeTitlePage(pdf *fpdf.Fpdf, opts Options, chapterCount int) {
pdf.AddPage()
pdf.SetY(opts.Page.H * 0.30)
titleSize := opts.FontSize * 1.9
pdf.SetFontSize(titleSize)
pdf.MultiCell(0, titleSize*1.35*pointsToMM, strings.ToUpper(opts.Title), "", "C", false)
pdf.Ln(opts.FontSize * pointsToMM * 2)
pdf.SetFontSize(opts.FontSize * 0.85)
pdf.SetTextColor(90, 90, 90)
pdf.MultiCell(0, opts.FontSize*1.4*pointsToMM,
fmt.Sprintf("%d chương", chapterCount), "", "C", false)
if opts.SourceURL != "" {
pdf.MultiCell(0, opts.FontSize*1.4*pointsToMM, opts.SourceURL, "", "C", false)
}
pdf.SetTextColor(0, 0, 0)
pdf.SetFontSize(opts.FontSize)
}
+79
View File
@@ -0,0 +1,79 @@
package pdf
import (
"os"
"path/filepath"
"testing"
)
func testOptions(t *testing.T) Options {
t.Helper()
return Options{
Page: PhonePage,
Margin: 6,
Font: BundledFont(),
FontSize: 12,
LineSpacing: 1.55,
Title: "TRỞ LẠI NĂM THÁNG CŨ",
SourceURL: "https://example.test/n.html",
}
}
func TestWriteProducesReadablePDF(t *testing.T) {
path := filepath.Join(t.TempDir(), "out.pdf")
chapters := []Chapter{
{Heading: "Chương 1", Paragraphs: []string{
"Nghe vị trưởng tử nói chuyện với nàng.",
"Một đoạn văn khác để kiểm tra ngắt dòng tự động trên trang nhỏ.",
}},
{Heading: "Chương 2", Paragraphs: []string{"Đoạn cuối."}},
}
if err := Write(path, testOptions(t), chapters); err != nil {
t.Fatalf("Write: %v", err)
}
info, err := os.Stat(path)
if err != nil {
t.Fatalf("stat output: %v", err)
}
if info.Size() == 0 {
t.Fatal("wrote an empty PDF")
}
header := make([]byte, 5)
f, err := os.Open(path)
if err != nil {
t.Fatal(err)
}
defer f.Close()
if _, err := f.Read(header); err != nil {
t.Fatal(err)
}
if string(header) != "%PDF-" {
t.Errorf("output does not start with a PDF header, got %q", header)
}
}
// A novel with many chapters must not overflow a page; auto page break plus the
// footer reserve handles that, so a long chapter should span several pages.
func TestWriteHandlesLongChapters(t *testing.T) {
path := filepath.Join(t.TempDir(), "long.pdf")
paragraphs := make([]string, 200)
for i := range paragraphs {
paragraphs[i] = "Một đoạn văn dài để buộc trình kết xuất phải sang trang mới nhiều lần."
}
if err := Write(path, testOptions(t), []Chapter{{Heading: "Chương 1", Paragraphs: paragraphs}}); err != nil {
t.Fatalf("Write: %v", err)
}
info, err := os.Stat(path)
if err != nil {
t.Fatal(err)
}
if info.Size() < 2000 {
t.Errorf("output suspiciously small (%d bytes) for 200 paragraphs", info.Size())
}
}
+2 -2
View File
@@ -12,8 +12,8 @@ import (
"github.com/go-telegram/bot"
"github.com/go-telegram/bot/models"
"github.com/tiennm99/monkeyd-crawler/export"
crawler "github.com/tiennm99/monkeyd-crawler/monkeyd"
"github.com/tiennm99/miti99bot/internal/modules/monkeyd/crawler"
"github.com/tiennm99/miti99bot/internal/modules/monkeyd/export"
"github.com/tiennm99/miti99bot/internal/log"
"github.com/tiennm99/miti99bot/internal/modules"