diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 145095a..7ba2432 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -16,12 +16,7 @@ jobs: matrix: go: ['1.26.5'] steps: - # The monkeyd module builds against third_party/monkeyd-crawler, which is - # a submodule wired in through a go.mod replace directive. Without it - # checked out, every Go step fails to resolve the package. - uses: actions/checkout@v6 - with: - submodules: true - uses: actions/setup-go@v6 with: diff --git a/.gitmodules b/.gitmodules deleted file mode 100644 index 2940d69..0000000 --- a/.gitmodules +++ /dev/null @@ -1,3 +0,0 @@ -[submodule "third_party/monkeyd-crawler"] - path = third_party/monkeyd-crawler - url = https://github.com/tiennm99/monkeyd-crawler.git diff --git a/.golangci.yml b/.golangci.yml index 375cbb1..61007a5 100644 --- a/.golangci.yml +++ b/.golangci.yml @@ -50,6 +50,9 @@ linters: - name: unused-parameter disabled: true exclusions: + # renderer/ is the Node service; its node_modules can carry stray Go files. + paths: + - renderer rules: # Tests routinely pass dummy values, swallow errors from helpers, and # use higher cyclomatic complexity in table-driven cases. Suppress the diff --git a/Dockerfile b/Dockerfile index 29d9534..d7055d4 100644 --- a/Dockerfile +++ b/Dockerfile @@ -1,12 +1,9 @@ FROM golang:1.26.5-alpine AS builder WORKDIR /src -# The monkeyd-crawler submodule is resolved through a `replace` directive, so -# its go.mod must be present before `go mod download` can read the build list. -# Only the module files are copied here, keeping this layer cached across -# ordinary source edits. +# Only the module files are copied first, keeping the download layer cached +# across ordinary source edits. COPY go.mod go.sum ./ -COPY third_party/monkeyd-crawler/go.mod third_party/monkeyd-crawler/go.sum ./third_party/monkeyd-crawler/ RUN go mod download COPY . . diff --git a/go.mod b/go.mod index 6b1d15d..c8e3893 100644 --- a/go.mod +++ b/go.mod @@ -3,22 +3,19 @@ module github.com/tiennm99/miti99bot go 1.26.5 require ( + github.com/go-pdf/fpdf v0.9.0 github.com/go-telegram/bot v1.20.0 github.com/ledongthuc/pdf v0.0.0-20260907135840-6c8c28e0e8a0 github.com/robfig/cron/v3 v3.0.1 github.com/testcontainers/testcontainers-go v0.43.0 github.com/testcontainers/testcontainers-go/modules/mongodb v0.43.0 - github.com/tiennm99/monkeyd-crawler v0.0.0 go.mongodb.org/mongo-driver/v2 v2.7.0 golang.org/x/image v0.45.0 + golang.org/x/net v0.57.0 + golang.org/x/sync v0.22.0 golang.org/x/text v0.41.0 ) -require ( - github.com/go-pdf/fpdf v0.9.0 // indirect - golang.org/x/net v0.57.0 // indirect -) - require ( dario.cat/mergo v1.0.2 // indirect github.com/Azure/go-ansiterm v0.0.0-20250102033503-faa5f7b0171c // indirect @@ -72,9 +69,6 @@ require ( go.opentelemetry.io/otel/metric v1.41.0 // indirect go.opentelemetry.io/otel/trace v1.41.0 // indirect golang.org/x/crypto v0.54.0 // indirect - golang.org/x/sync v0.22.0 // indirect golang.org/x/sys v0.47.0 // indirect gopkg.in/yaml.v3 v3.0.1 // indirect ) - -replace github.com/tiennm99/monkeyd-crawler => ./third_party/monkeyd-crawler diff --git a/internal/modules/monkeyd/crawler/chapter.go b/internal/modules/monkeyd/crawler/chapter.go new file mode 100644 index 0000000..bb12a1f --- /dev/null +++ b/internal/modules/monkeyd/crawler/chapter.go @@ -0,0 +1,187 @@ +package crawler + +import ( + "bytes" + "fmt" + "strconv" + "strings" + + "golang.org/x/net/html" +) + +// contentElementID is the container holding a chapter's body text. +const contentElementID = "chapter-content-render" + +// Chapter is one fetched chapter, reduced to plain paragraphs. +type Chapter struct { + Label string + URL string + Paragraphs []string +} + +// Heading is the chapter title to print. Labels are often a bare number, which +// reads poorly as a heading, so those get the Vietnamese word for "chapter". +func (c *Chapter) Heading() string { + label := strings.TrimSpace(c.Label) + if label == "" { + return "Chương" + } + if _, err := strconv.Atoi(label); err == nil { + return "Chương " + label + } + return label +} + +// WordCount is a rough word count, used to sanity-check extraction. +func (c *Chapter) WordCount() int { + n := 0 + for _, p := range c.Paragraphs { + n += len(strings.Fields(p)) + } + return n +} + +// blockTags end the current paragraph when opened or closed. +var blockTags = map[string]bool{ + "p": true, "div": true, "br": true, "hr": true, "blockquote": true, + "h1": true, "h2": true, "h3": true, "h4": true, "h5": true, "h6": true, + "li": true, "ul": true, "ol": true, "tr": true, +} + +// skipTags never contribute prose. +// +// Anchors are included because inside a chapter body they are always site +// chrome: the prev/next chapter navigation and the sponsor call-to-action are +// both links, while novel prose never needs one. Inline emphasis tags (b, i, +// em) are deliberately absent so italics in the prose survive; the site's icons +// use but carry no text. +var skipTags = map[string]bool{ + "script": true, "style": true, "noscript": true, "iframe": true, + "ins": true, "form": true, "select": true, "button": true, "textarea": true, + "a": true, "img": true, "svg": true, +} + +// junkClasses marks containers the site injects into the chapter body. Their +// whole subtree is dropped. +// +// These are matched on class rather than position because the blocks move: the +// sponsor block opens the body on most chapters but is absent on others, and +// the watermark is planted at a different paragraph in every chapter. Only +// site-specific class names are listed; generic Bootstrap utilities such as +// "my-4" or "text-center" are not, since prose could legitimately carry them. +// +// Note the sibling class "actac" is NOT junk: it wraps the real chapter text and +// carries style="display:none", because the site gates the body behind a click +// on the sponsor link and reveals it with JavaScript. Skipping hidden elements, +// or skipping "act*" as a family, would therefore discard the whole chapter. +var junkClasses = map[string]bool{ + "actcl": true, // sponsor block shown in place of the gated chapter body + "signature": true, // "[Truyện được đăng tải duy nhất tại ...]" source watermark +} + +// hasJunkClass reports whether a node is an injected non-prose container. +func hasJunkClass(n *html.Node) bool { + for _, tok := range strings.Fields(attr(n, "class")) { + if junkClasses[tok] { + return true + } + } + return false +} + +// ParseChapter extracts a chapter's paragraphs, restoring the words the site +// serves through CSS :before rules instead of markup. +func ParseChapter(page []byte, ref ChapterRef) (*Chapter, error) { + doc, err := html.Parse(bytes.NewReader(page)) + if err != nil { + return nil, fmt.Errorf("parse chapter %s: %w", ref.URL, err) + } + content := elementByID(doc, contentElementID) + if content == nil { + return nil, fmt.Errorf("chapter %s: no #%s container (page layout may have changed)", + ref.URL, contentElementID) + } + + ch := &Chapter{ + Label: ref.Label, + URL: ref.URL, + Paragraphs: extractParagraphs(content, ParseWordClasses(page)), + } + if len(ch.Paragraphs) == 0 { + return nil, fmt.Errorf("chapter %s: extracted no text", ref.URL) + } + ch.Paragraphs = dropRepeatedTitle(ch.Paragraphs, ref.Label) + return ch, nil +} + +// extractParagraphs walks the content subtree into plain paragraphs, replacing +// each word-carrying element with the word its CSS rule injects. +func extractParagraphs(content *html.Node, words map[string]string) []string { + var b strings.Builder + + var walk func(*html.Node) + walk = func(n *html.Node) { + switch n.Type { + case html.TextNode: + // The HTML parser has already decoded entities such as ư. + b.WriteString(n.Data) + return + case html.ElementNode: + if skipTags[n.Data] || hasJunkClass(n) { + return + } + // These elements are empty in the markup; the CSS word replaces them. + if word, ok := injectedWord(n, words); ok { + b.WriteString(word) + return + } + if blockTags[n.Data] { + b.WriteByte('\n') + } + } + + for c := n.FirstChild; c != nil; c = c.NextSibling { + walk(c) + } + + if n.Type == html.ElementNode && blockTags[n.Data] { + b.WriteByte('\n') + } + } + walk(content) + + var paragraphs []string + for _, line := range strings.Split(b.String(), "\n") { + if p := collapseSpaces(line); p != "" { + paragraphs = append(paragraphs, p) + } + } + return paragraphs +} + +// injectedWord returns the word a node's class supplies via CSS, if any. +func injectedWord(n *html.Node, words map[string]string) (string, bool) { + class := attr(n, "class") + if class == "" { + return "", false + } + for _, tok := range strings.Fields(class) { + if word, ok := words[tok]; ok { + return word, true + } + } + return "", false +} + +// dropRepeatedTitle removes a leading paragraph that only repeats the chapter +// label, since the export prints its own heading. +func dropRepeatedTitle(paragraphs []string, label string) []string { + if len(paragraphs) < 2 { + return paragraphs + } + first := strings.TrimSpace(paragraphs[0]) + if strings.EqualFold(first, strings.TrimSpace(label)) { + return paragraphs[1:] + } + return paragraphs +} diff --git a/internal/modules/monkeyd/crawler/chapter_test.go b/internal/modules/monkeyd/crawler/chapter_test.go new file mode 100644 index 0000000..f366951 --- /dev/null +++ b/internal/modules/monkeyd/crawler/chapter_test.go @@ -0,0 +1,200 @@ +package crawler + +import ( + "strings" + "testing" +) + +// chapterFixture mirrors the real page shape: words split between markup and +// CSS :before rules, HTML entities,   spacer paragraphs and an ad script +// inside the content container. +const chapterFixture = ` + +

TEN TRUYEN - 1

+
+

1

+

 

+

Nghe trưởng tử noi.

+

 

+

Một và di.

+ +
` + +func TestParseWordClassesDecodesEscapes(t *testing.T) { + words := ParseWordClasses([]byte(chapterFixture)) + + for class, want := range map[string]string{ + "t-aaa": "vị", + "j-bbb": "trồ", + "z-ccc": "nàng", + } { + if got := words[class]; got != want { + t.Errorf("words[%q] = %q, want %q", class, got, want) + } + } + if len(words) != 4 { + t.Errorf("got %d rules, want 4", len(words)) + } +} + +func TestParseChapterRestoresCSSWords(t *testing.T) { + ch, err := ParseChapter([]byte(chapterFixture), ChapterRef{Label: "1", URL: "http://x/1.html"}) + if err != nil { + t.Fatalf("ParseChapter: %v", err) + } + + want := []string{ + "Nghe vị trưởng tử noi.", + "Một trồ và nàng di.", + } + if len(ch.Paragraphs) != len(want) { + t.Fatalf("got %d paragraphs %q, want %d", len(ch.Paragraphs), ch.Paragraphs, len(want)) + } + for i, w := range want { + if ch.Paragraphs[i] != w { + t.Errorf("paragraph %d = %q, want %q", i, ch.Paragraphs[i], w) + } + } +} + +// The CSS-injected words are the difference between real text and text with +// silent holes, so guard against a regression that drops them. +func TestParseChapterWithoutCSSWouldLoseWords(t *testing.T) { + withoutCSS := strings.Replace(chapterFixture, `.t-aaa:before { content: "v\1ecb "; }`, "", 1) + + ch, err := ParseChapter([]byte(withoutCSS), ChapterRef{Label: "1", URL: "http://x/1.html"}) + if err != nil { + t.Fatalf("ParseChapter: %v", err) + } + if strings.Contains(ch.Paragraphs[0], "vị") { + t.Fatal("word appeared without its CSS rule; fixture no longer proves anything") + } + if want := "Nghe trưởng tử noi."; ch.Paragraphs[0] != want { + t.Errorf("paragraph 0 = %q, want %q", ch.Paragraphs[0], want) + } +} + +func TestParseChapterDropsScriptsAndSpacers(t *testing.T) { + ch, err := ParseChapter([]byte(chapterFixture), ChapterRef{Label: "1", URL: "http://x/1.html"}) + if err != nil { + t.Fatalf("ParseChapter: %v", err) + } + for _, p := range ch.Paragraphs { + if strings.Contains(p, "ads()") { + t.Errorf("script text leaked into paragraph %q", p) + } + if strings.TrimSpace(p) == "" { + t.Error("empty spacer paragraph was kept") + } + if strings.Contains(p, " ") { + t.Errorf("non-breaking space survived in %q", p) + } + } +} + +// The leading "

1

" repeats the chapter label and would print twice. +func TestParseChapterDropsRepeatedTitle(t *testing.T) { + ch, err := ParseChapter([]byte(chapterFixture), ChapterRef{Label: "1", URL: "http://x/1.html"}) + if err != nil { + t.Fatalf("ParseChapter: %v", err) + } + if ch.Paragraphs[0] == "1" { + t.Error("repeated chapter label was kept as a paragraph") + } +} + +// gatedChapterFixture mirrors how most chapters are served: a visible sponsor +// block (div.actcl) stands in for the body, while the real text sits in a +// sibling div.actac hidden with display:none and revealed by the site's +// JavaScript. A source watermark and prev/next navigation bracket the prose. +const gatedChapterFixture = ` + +
+
+

Moi Quy doc gia CLICK vao lien ket

+

mo ung dung Shopee, sau do quay tro lai de doc!

+ + CLICK +

MonkeyD va doi ngu Editor xin chan thanh cam on!

+
+
+

Doan van dau tien co tri.

+

 

+

[Truyen duoc dang tai duy nhat tai MonkeyDD.com - https://monkeydd.com/n/10.html.]

+

Doan van cuoi cung.

+
+ +
` + +// The gated body is the one thing that must survive: it is hidden with +// display:none, so any rule that drops hidden or "act*" containers silently +// discards the entire chapter. +func TestParseChapterKeepsGatedBody(t *testing.T) { + ch, err := ParseChapter([]byte(gatedChapterFixture), ChapterRef{Label: "10", URL: "http://x/10.html"}) + if err != nil { + t.Fatalf("ParseChapter: %v", err) + } + + want := []string{ + "Doan van dau tien co vị tri.", + "Doan van cuoi cung.", + } + if len(ch.Paragraphs) != len(want) { + t.Fatalf("got %d paragraphs %q, want %d", len(ch.Paragraphs), ch.Paragraphs, len(want)) + } + for i, w := range want { + if ch.Paragraphs[i] != w { + t.Errorf("paragraph %d = %q, want %q", i, ch.Paragraphs[i], w) + } + } +} + +// Everything the site injects around the prose must be gone. +func TestParseChapterDropsInjectedBlocks(t *testing.T) { + ch, err := ParseChapter([]byte(gatedChapterFixture), ChapterRef{Label: "10", URL: "http://x/10.html"}) + if err != nil { + t.Fatalf("ParseChapter: %v", err) + } + body := strings.Join(ch.Paragraphs, "\n") + + for _, junk := range []string{ + "Shopee", // sponsor copy + "CLICK", // sponsor call to action + "cam on", // sponsor sign-off + "MonkeyDD.com", // source watermark + "Chương trước", // navigation + "Chương sau", // navigation + } { + if strings.Contains(body, junk) { + t.Errorf("injected text %q survived extraction in %q", junk, body) + } + } +} + +func TestParseChapterMissingContainer(t *testing.T) { + if _, err := ParseChapter([]byte(`

hi

`), + ChapterRef{URL: "http://x/1.html"}); err == nil { + t.Fatal("want an error when the content container is absent") + } +} + +func TestChapterHeading(t *testing.T) { + for _, tc := range []struct{ label, want string }{ + {"14", "Chương 14"}, + {"Chương 12", "Chương 12"}, + {"", "Chương"}, + } { + ch := &Chapter{Label: tc.label} + if got := ch.Heading(); got != tc.want { + t.Errorf("Heading(%q) = %q, want %q", tc.label, got, tc.want) + } + } +} diff --git a/internal/modules/monkeyd/crawler/client.go b/internal/modules/monkeyd/crawler/client.go new file mode 100644 index 0000000..a0cdb16 --- /dev/null +++ b/internal/modules/monkeyd/crawler/client.go @@ -0,0 +1,135 @@ +package crawler + +import ( + "context" + "errors" + "fmt" + "io" + "net/http" + "sync" + "time" +) + +// The site rejects requests without a browser-like User-Agent. +const defaultUserAgent = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 " + + "(KHTML, like Gecko) Chrome/124.0.0.0 Safari/537.36" + +// maxPageSize caps how much of a response we buffer; chapter pages are ~130 KB. +const maxPageSize = 8 << 20 + +// statusError reports an unexpected HTTP status so Get can decide whether +// retrying is worthwhile. +type statusError struct { + code int + status string +} + +func (e *statusError) Error() string { return "unexpected status " + e.status } + +// retryable is true for transient failures. A 404 or 403 will not fix itself, +// so those fail immediately instead of burning the retry budget. +func (e *statusError) retryable() bool { + return e.code == http.StatusTooManyRequests || e.code >= 500 +} + +// Client fetches pages from monkeydd.com. It spaces requests out by a fixed +// delay no matter how many goroutines call Get, so raising the worker count +// never raises the request rate, and it retries transient failures with +// exponential backoff. +type Client struct { + http *http.Client + ua string + delay time.Duration + retries int + + mu sync.Mutex + nextSlot time.Time +} + +func NewClient(delay time.Duration, retries int) *Client { + return &Client{ + http: &http.Client{Timeout: 45 * time.Second}, + ua: defaultUserAgent, + delay: delay, + retries: retries, + } +} + +// reserve claims the next request slot and blocks until it comes due, holding +// the global rate at one request per delay across all callers. +func (c *Client) reserve(ctx context.Context) error { + c.mu.Lock() + slot := c.nextSlot + if now := time.Now(); slot.Before(now) { + slot = now + } + c.nextSlot = slot.Add(c.delay) + c.mu.Unlock() + + wait := time.Until(slot) + if wait <= 0 { + return nil + } + timer := time.NewTimer(wait) + defer timer.Stop() + select { + case <-ctx.Done(): + return ctx.Err() + case <-timer.C: + return nil + } +} + +// Get fetches url, retrying transient failures. +func (c *Client) Get(ctx context.Context, url string) ([]byte, error) { + var lastErr error + for attempt := 0; attempt <= c.retries; attempt++ { + if attempt > 0 { + backoff := time.Duration(1< 0 { + c.logf("warning: %d chapter(s) appear only in the chapter dropdown and were not "+ + "in the landing page list; verify the export is complete", len(extra)) + } + if len(final) != len(novel.Chapters) { + c.logf("chapter list reconciled to %d chapters using the in-chapter dropdown", len(final)) + } + novel.Chapters = final + return novel, nil +} + +// NovelInfo fetches only the landing page and returns what it carries: title, +// slug, tags, and the chapter list as that page shows it. +// +// Unlike Novel it does not also fetch a chapter page to cross-check the chapter +// list, so it costs a single request. Callers that only want metadata should +// prefer it; callers about to export every chapter want Novel, whose +// reconciliation guards against a silently truncated list. +func (c *Crawler) NovelInfo(ctx context.Context, novelURL string) (*Novel, error) { + page, err := c.page(ctx, novelURL) + if err != nil { + return nil, err + } + return ParseNovelPage(page, novelURL) +} + +// Chapters fetches every chapter concurrently and returns them in reading +// order. Any chapter that cannot be fetched or parsed fails the whole run +// rather than yielding a book with a hole in it. +func (c *Crawler) Chapters(ctx context.Context, novel *Novel) ([]*Chapter, error) { + chapters := make([]*Chapter, len(novel.Chapters)) + + workers := c.Workers + if workers < 1 { + workers = 1 + } + + group, groupCtx := errgroup.WithContext(ctx) + group.SetLimit(workers) + + var mu sync.Mutex + done := 0 + + for i, ref := range novel.Chapters { + i, ref := i, ref + group.Go(func() error { + page, err := c.page(groupCtx, ref.URL) + if err != nil { + return err + } + chapter, err := ParseChapter(page, ref) + if err != nil { + return err + } + chapters[i] = chapter + + mu.Lock() + done++ + c.logf("fetched %d/%d: %s (%d words)", done, len(novel.Chapters), + chapter.Heading(), chapter.WordCount()) + mu.Unlock() + return nil + }) + } + + if err := group.Wait(); err != nil { + return nil, err + } + return chapters, nil +} + +// page returns a page from the cache when available, otherwise fetches and +// caches it. +func (c *Crawler) page(ctx context.Context, pageURL string) ([]byte, error) { + path := c.cachePath(pageURL) + if path != "" { + if body, err := os.ReadFile(path); err == nil && len(body) > 0 { //nolint:gosec // G304: cachePath strips every path separator from the URL + return body, nil + } + } + + body, err := c.Client.Get(ctx, pageURL) + if err != nil { + return nil, err + } + + if path != "" { + if err := os.MkdirAll(filepath.Dir(path), 0o750); err == nil { + // A failed cache write must not fail the crawl. + _ = os.WriteFile(path, body, 0o600) + } + } + return body, nil +} + +// unsafeFileChars matches everything not allowed in a cache file name. +var unsafeFileChars = regexp.MustCompile(`[^A-Za-z0-9._-]+`) + +// cachePath maps a page URL to a cache file, or "" when caching is disabled. +func (c *Crawler) cachePath(pageURL string) string { + if c.CacheDir == "" { + return "" + } + u, err := url.Parse(pageURL) + if err != nil { + return "" + } + name := unsafeFileChars.ReplaceAllString(strings.Trim(u.Path, "/"), "_") + if name == "" { + return "" + } + if !strings.HasSuffix(name, ".html") { + name += ".html" + } + return filepath.Join(c.CacheDir, name) +} + +// TotalWords sums the word count across chapters. +func TotalWords(chapters []*Chapter) int { + n := 0 + for _, ch := range chapters { + n += ch.WordCount() + } + return n +} + +// Describe renders a one-line summary of a crawl result. +func Describe(novel *Novel, chapters []*Chapter) string { + return fmt.Sprintf("%s — %d chapters, %d words", novel.Title, len(chapters), TotalWords(chapters)) +} diff --git a/internal/modules/monkeyd/crawler/css_words.go b/internal/modules/monkeyd/crawler/css_words.go new file mode 100644 index 0000000..7566e61 --- /dev/null +++ b/internal/modules/monkeyd/crawler/css_words.go @@ -0,0 +1,50 @@ +package crawler + +import ( + "regexp" + "strconv" + "strings" +) + +// The site hides part of every chapter behind CSS rather than putting it in the +// markup. Chapter HTML carries empty elements such as +// +// Nghe trưởng tử +// +// and the page stylesheet supplies the missing word: +// +// .t-3e625e...:before { content: "vị"; } +// +// Reading DOM text alone therefore drops hundreds of words per chapter without +// any visible error. wordRule finds those rules so the words can be put back. +var wordRule = regexp.MustCompile(`\.([A-Za-z0-9_-]+)\s*::?before\s*\{[^}]*?content\s*:\s*"((?:[^"\\]|\\.)*)"`) + +// cssEscape matches a CSS character escape: a hex code point, optionally +// followed by one whitespace terminator, or an escaped literal character. +var cssEscape = regexp.MustCompile(`\\([0-9A-Fa-f]{1,6})\s?|\\(.)`) + +// ParseWordClasses maps CSS class name to the word its :before rule injects. +func ParseWordClasses(page []byte) map[string]string { + words := make(map[string]string) + for _, m := range wordRule.FindAllSubmatch(page, -1) { + words[string(m[1])] = decodeCSSString(string(m[2])) + } + return words +} + +// decodeCSSString resolves the escape sequences allowed inside a CSS string. +func decodeCSSString(s string) string { + if !strings.Contains(s, `\`) { + return s + } + return cssEscape.ReplaceAllStringFunc(s, func(esc string) string { + m := cssEscape.FindStringSubmatch(esc) + if m[1] != "" { + if cp, err := strconv.ParseInt(m[1], 16, 32); err == nil && cp > 0 { + return string(rune(cp)) + } + return "" + } + return m[2] + }) +} diff --git a/internal/modules/monkeyd/crawler/html_nodes.go b/internal/modules/monkeyd/crawler/html_nodes.go new file mode 100644 index 0000000..2aee719 --- /dev/null +++ b/internal/modules/monkeyd/crawler/html_nodes.go @@ -0,0 +1,95 @@ +package crawler + +import ( + "strings" + + "golang.org/x/net/html" +) + +// attr returns the value of the named attribute, or "" when absent. +func attr(n *html.Node, name string) string { + for _, a := range n.Attr { + if a.Key == name { + return a.Val + } + } + return "" +} + +// hasClass reports whether the node carries the given class token. +func hasClass(n *html.Node, class string) bool { + for _, tok := range strings.Fields(attr(n, "class")) { + if tok == class { + return true + } + } + return false +} + +// findNode returns the first node in document order satisfying match. +func findNode(root *html.Node, match func(*html.Node) bool) *html.Node { + if match(root) { + return root + } + for c := root.FirstChild; c != nil; c = c.NextSibling { + if found := findNode(c, match); found != nil { + return found + } + } + return nil +} + +// findAllNodes returns every node satisfying match, in document order. +func findAllNodes(root *html.Node, match func(*html.Node) bool) []*html.Node { + var out []*html.Node + var walk func(*html.Node) + walk = func(n *html.Node) { + if match(n) { + out = append(out, n) + } + for c := n.FirstChild; c != nil; c = c.NextSibling { + walk(c) + } + } + walk(root) + return out +} + +// elementByID finds an element by its id attribute. +func elementByID(root *html.Node, id string) *html.Node { + return findNode(root, func(n *html.Node) bool { + return n.Type == html.ElementNode && attr(n, "id") == id + }) +} + +// elementByTag finds the first element with the given tag name. +func elementByTag(root *html.Node, tag string) *html.Node { + return findNode(root, func(n *html.Node) bool { + return n.Type == html.ElementNode && n.Data == tag + }) +} + +// nodeText collects the descendant text of a node with whitespace collapsed. +func nodeText(n *html.Node) string { + if n == nil { + return "" + } + var b strings.Builder + var walk func(*html.Node) + walk = func(n *html.Node) { + if n.Type == html.TextNode { + b.WriteString(n.Data) + } + for c := n.FirstChild; c != nil; c = c.NextSibling { + walk(c) + } + } + walk(n) + return collapseSpaces(b.String()) +} + +// collapseSpaces trims the string and reduces every whitespace run, including +// the non-breaking spaces the site emits as  , to a single space. +func collapseSpaces(s string) string { + return strings.Join(strings.Fields(s), " ") +} diff --git a/internal/modules/monkeyd/crawler/novel.go b/internal/modules/monkeyd/crawler/novel.go new file mode 100644 index 0000000..e097dfd --- /dev/null +++ b/internal/modules/monkeyd/crawler/novel.go @@ -0,0 +1,223 @@ +package crawler + +import ( + "bytes" + "fmt" + "net/url" + "strings" + + "golang.org/x/net/html" +) + +// ChapterRef points at a single chapter listed on a novel page. +type ChapterRef struct { + Label string // as shown on the site, e.g. "14" or "Chương 12" + URL string +} + +// Novel is a novel's landing page: its title and its chapters in reading order. +type Novel struct { + Title string + Slug string + URL string + + // Tags are the novel's own genres, as labelled on the site, in the order + // they appear. Empty when the page lists none. + Tags []string + Chapters []ChapterRef +} + +// ParseNovelPage reads the title and chapter list from a novel landing page. +// +// Chapter URLs are always taken from the anchors on the page. Slugs are not +// uniform across novels ("/14.html" on one, "/chuong-12.html" on another) and +// numbering has gaps, so generating URLs from a chapter count would fetch 404s +// and miss real chapters. +func ParseNovelPage(page []byte, pageURL string) (*Novel, error) { + doc, err := html.Parse(bytes.NewReader(page)) + if err != nil { + return nil, fmt.Errorf("parse novel page: %w", err) + } + base, err := url.Parse(pageURL) + if err != nil { + return nil, fmt.Errorf("parse novel url: %w", err) + } + + novel := &Novel{ + Title: novelTitle(doc), + Slug: slugFromNovelURL(base), + URL: pageURL, + Tags: tagsFromInfo(doc), + Chapters: chapterRefsFromList(doc, base), + } + if novel.Title == "" { + novel.Title = novel.Slug + } + if len(novel.Chapters) == 0 { + return nil, fmt.Errorf("no chapters found on %s (page layout may have changed)", pageURL) + } + return novel, nil +} + +// novelTitle prefers the

heading and falls back to the document title. +func novelTitle(doc *html.Node) string { + if h1 := elementByTag(doc, "h1"); h1 != nil { + if t := nodeText(h1); t != "" { + return t + } + } + return nodeText(elementByTag(doc, "title")) +} + +// tagsFromInfo reads the novel's own genres out of the info block. +// +// The anchors are matched on the schema.org itemprop="genre" microdata rather +// than on their href. The page also carries a site-wide genre menu linking every +// category — 69 distinct ones against this novel's 6 on a sampled page — so +// matching the /the-loai/ URL shape would pull in the whole navigation. +func tagsFromInfo(doc *html.Node) []string { + links := findAllNodes(doc, func(n *html.Node) bool { + return n.Type == html.ElementNode && n.Data == "a" && attr(n, "itemprop") == "genre" + }) + + var tags []string + seen := make(map[string]bool, len(links)) + for _, link := range links { + // The label appears both as the link text and as its title attribute; + // prefer the text and fall back to the attribute. + name := nodeText(link) + if name == "" { + name = collapseSpaces(attr(link, "title")) + } + if name == "" || seen[name] { + continue + } + seen[name] = true + tags = append(tags, name) + } + return tags +} + +// slugFromNovelURL turns https://host/tro-lai-nam-thang-cu.html into +// "tro-lai-nam-thang-cu". +func slugFromNovelURL(u *url.URL) string { + seg := strings.Trim(u.Path, "/") + if i := strings.LastIndex(seg, "/"); i >= 0 { + seg = seg[i+1:] + } + return strings.TrimSuffix(seg, ".html") +} + +// chapterRefsFromList reads the "list-chapters" block on the landing page. +// The site lists newest first, so the result is reversed into reading order. +func chapterRefsFromList(doc *html.Node, base *url.URL) []ChapterRef { + list := findNode(doc, func(n *html.Node) bool { + return n.Type == html.ElementNode && hasClass(n, "list-chapters") + }) + if list == nil { + return nil + } + + titles := findAllNodes(list, func(n *html.Node) bool { + return n.Type == html.ElementNode && hasClass(n, "episode-title") + }) + + var refs []ChapterRef + for _, title := range titles { + link := elementByTag(title, "a") + if link == nil { + continue + } + href := strings.TrimSpace(attr(link, "href")) + if href == "" { + continue + } + abs, err := base.Parse(href) + if err != nil { + continue + } + refs = append(refs, ChapterRef{Label: nodeText(link), URL: abs.String()}) + } + return reverseRefs(refs) +} + +// ChapterRefsFromSelect reads the chapter dropdown embedded in every chapter +// page, whose options hold "novel-slug,chapter-slug" pairs. This is a second, +// independent view of the chapter list used to cross-check the landing page. +func ChapterRefsFromSelect(page []byte, base *url.URL) ([]ChapterRef, error) { + doc, err := html.Parse(bytes.NewReader(page)) + if err != nil { + return nil, fmt.Errorf("parse chapter page: %w", err) + } + sel := elementByID(doc, "selected_chapter") + if sel == nil { + return nil, nil + } + + var refs []ChapterRef + for _, opt := range findAllNodes(sel, func(n *html.Node) bool { + return n.Type == html.ElementNode && n.Data == "option" + }) { + novelSlug, chapterSlug, ok := strings.Cut(attr(opt, "value"), ",") + if !ok || novelSlug == "" || chapterSlug == "" { + continue + } + abs, err := base.Parse("/" + novelSlug + "/" + chapterSlug + ".html") + if err != nil { + continue + } + refs = append(refs, ChapterRef{Label: nodeText(opt), URL: abs.String()}) + } + return reverseRefs(refs), nil +} + +// ReconcileChapterRefs picks the chapter list to crawl from the landing page +// list and the in-chapter dropdown. +// +// Both views come from the same site ordering, so when the dropdown is a +// superset it is preferred: that keeps the run correct even if the landing page +// ever truncates or paginates its list. Anything the dropdown alone knows about +// while disagreeing on order is returned as extra so the caller can warn rather +// than silently export a short book. +func ReconcileChapterRefs(fromList, fromSelect []ChapterRef) (final, extra []ChapterRef) { + if len(fromSelect) == 0 { + return fromList, nil + } + + inSelect := refURLSet(fromSelect) + if listIsSubsetOf(fromList, inSelect) { + return fromSelect, nil + } + + inList := refURLSet(fromList) + for _, ref := range fromSelect { + if !inList[ref.URL] { + extra = append(extra, ref) + } + } + return fromList, extra +} + +func refURLSet(refs []ChapterRef) map[string]bool { + set := make(map[string]bool, len(refs)) + for _, r := range refs { + set[r.URL] = true + } + return set +} + +func listIsSubsetOf(refs []ChapterRef, set map[string]bool) bool { + for _, r := range refs { + if !set[r.URL] { + return false + } + } + return true +} + +func reverseRefs(refs []ChapterRef) []ChapterRef { + for i, j := 0, len(refs)-1; i < j; i, j = i+1, j-1 { + refs[i], refs[j] = refs[j], refs[i] + } + return refs +} diff --git a/internal/modules/monkeyd/crawler/novel_test.go b/internal/modules/monkeyd/crawler/novel_test.go new file mode 100644 index 0000000..0e1cfa9 --- /dev/null +++ b/internal/modules/monkeyd/crawler/novel_test.go @@ -0,0 +1,170 @@ +package crawler + +import ( + "net/url" + "testing" +) + +// novelFixture reproduces the two traits that break naive crawlers: the list is +// newest-first, and chapter numbering has a gap (no chapter 4). +const novelFixture = `TEN TRUYEN +

TEN TRUYEN

+
+ + + + +
` + +func TestParseNovelPageOrdersChaptersForReading(t *testing.T) { + novel, err := ParseNovelPage([]byte(novelFixture), "https://monkeydd.com/n.html") + if err != nil { + t.Fatalf("ParseNovelPage: %v", err) + } + + if novel.Title != "TEN TRUYEN" { + t.Errorf("Title = %q", novel.Title) + } + if novel.Slug != "n" { + t.Errorf("Slug = %q, want %q", novel.Slug, "n") + } + + wantLabels := []string{"1", "2", "3", "5"} + if len(novel.Chapters) != len(wantLabels) { + t.Fatalf("got %d chapters, want %d", len(novel.Chapters), len(wantLabels)) + } + for i, want := range wantLabels { + if novel.Chapters[i].Label != want { + t.Errorf("chapter %d label = %q, want %q", i, novel.Chapters[i].Label, want) + } + } + // Relative hrefs must resolve against the novel URL. + if got, want := novel.Chapters[1].URL, "https://monkeydd.com/n/2.html"; got != want { + t.Errorf("chapter 2 URL = %q, want %q", got, want) + } +} + +// tagsFixture mirrors the real page: the novel's own genres carry +// itemprop="genre", while a site-wide menu links every category without it. A +// parser matching on the /the-loai/ href shape would swallow the whole menu. +const tagsFixture = `TEN TRUYEN +

TEN TRUYEN

+
+
Thể loại
+
+ Trọng Sinh + Cổ Đại + + Cổ Đại +
+
+ +
+ +
` + +func TestParseNovelPageReadsTags(t *testing.T) { + novel, err := ParseNovelPage([]byte(tagsFixture), "https://monkeydd.com/n.html") + if err != nil { + t.Fatalf("ParseNovelPage: %v", err) + } + + // "Cổ Đại" has its inner whitespace collapsed, the empty anchor falls back + // to its title attribute, and the repeated genre appears once. + want := []string{"Trọng Sinh", "Cổ Đại", "Gia Đình"} + if len(novel.Tags) != len(want) { + t.Fatalf("Tags = %q, want %q", novel.Tags, want) + } + for i := range want { + if novel.Tags[i] != want[i] { + t.Errorf("Tags[%d] = %q, want %q", i, novel.Tags[i], want[i]) + } + } +} + +func TestParseNovelPageWithoutTags(t *testing.T) { + novel, err := ParseNovelPage([]byte(novelFixture), "https://monkeydd.com/n.html") + if err != nil { + t.Fatalf("ParseNovelPage: %v", err) + } + if len(novel.Tags) != 0 { + t.Errorf("Tags = %q, want none", novel.Tags) + } +} + +func TestParseNovelPageNoChapters(t *testing.T) { + if _, err := ParseNovelPage([]byte(`x`), + "https://monkeydd.com/n.html"); err == nil { + t.Fatal("want an error when no chapters are listed") + } +} + +const selectFixture = ` +` + +func TestChapterRefsFromSelect(t *testing.T) { + base, err := url.Parse("https://monkeydd.com/n.html") + if err != nil { + t.Fatal(err) + } + refs, err := ChapterRefsFromSelect([]byte(selectFixture), base) + if err != nil { + t.Fatalf("ChapterRefsFromSelect: %v", err) + } + + if len(refs) != 3 { + t.Fatalf("got %d refs %+v, want 3 (malformed option skipped)", len(refs), refs) + } + if got, want := refs[0].URL, "https://monkeydd.com/n/1.html"; got != want { + t.Errorf("first ref URL = %q, want %q", got, want) + } + if got, want := refs[1].URL, "https://monkeydd.com/n/chuong-3.html"; got != want { + t.Errorf("second ref URL = %q, want %q", got, want) + } +} + +func TestReconcileChapterRefs(t *testing.T) { + ref := func(u string) ChapterRef { return ChapterRef{Label: u, URL: u} } + + t.Run("dropdown superset wins", func(t *testing.T) { + list := []ChapterRef{ref("a"), ref("b")} + sel := []ChapterRef{ref("a"), ref("b"), ref("c")} + + final, extra := ReconcileChapterRefs(list, sel) + if len(final) != 3 { + t.Errorf("got %d chapters, want the 3 from the dropdown", len(final)) + } + if len(extra) != 0 { + t.Errorf("got %d extra, want 0", len(extra)) + } + }) + + t.Run("disagreement is reported not silently merged", func(t *testing.T) { + list := []ChapterRef{ref("a"), ref("z")} + sel := []ChapterRef{ref("a"), ref("b")} + + final, extra := ReconcileChapterRefs(list, sel) + if len(final) != 2 || final[1].URL != "z" { + t.Errorf("final = %+v, want the landing page list", final) + } + if len(extra) != 1 || extra[0].URL != "b" { + t.Errorf("extra = %+v, want [b] so the caller can warn", extra) + } + }) + + t.Run("empty dropdown falls back to the list", func(t *testing.T) { + list := []ChapterRef{ref("a")} + final, extra := ReconcileChapterRefs(list, nil) + if len(final) != 1 || len(extra) != 0 { + t.Errorf("final = %+v, extra = %+v", final, extra) + } + }) +} diff --git a/internal/modules/monkeyd/export/export.go b/internal/modules/monkeyd/export/export.go new file mode 100644 index 0000000..847cbc8 --- /dev/null +++ b/internal/modules/monkeyd/export/export.go @@ -0,0 +1,154 @@ +// Package export turns a novel URL into a PDF file: resolve the chapter list, +// fetch the chapters, render the PDF with the bundled font. +package export + +import ( + "context" + "fmt" + "net/url" + "path/filepath" + "regexp" + "strings" + "time" + + "github.com/tiennm99/miti99bot/internal/modules/monkeyd/crawler" + "github.com/tiennm99/miti99bot/internal/modules/monkeyd/pdf" +) + +// Defaults the bot advertises to users or reuses in other crawls. +const ( + DefaultFontSize = 10.0 + DefaultDelay = 400 * time.Millisecond +) + +// Fixed layout and crawl settings. The bot exposes only the font size. +const ( + lineSpacing = 1.55 // multiple of font size + margin = 6.0 // millimetres + workers = 4 + retries = 3 +) + +// Request describes one export. +type Request struct { + NovelURL string + + // OutDir receives the PDF, named after the novel title. + OutDir string + + // FontSize is in points. 0 means DefaultFontSize. + FontSize float64 + + // CacheDir stores raw pages so a re-export costs no requests. Empty + // disables the cache. + CacheDir string + + // Log receives progress messages. Optional. + Log func(format string, args ...any) +} + +// Result reports what was produced. +type Result struct { + Path string + Title string + SourceURL string + Chapters int + Words int + Page pdf.PageSize +} + +// Summary renders a one-line description of the exported book. +func (r *Result) Summary() string { + return fmt.Sprintf("%s — %d chapters, %d words", r.Title, r.Chapters, r.Words) +} + +// Export fetches the novel at req.NovelURL and writes it as a PDF, returning +// where it landed. The context bounds the whole crawl; cancelling it abandons +// the run without leaving a partial PDF behind. +func Export(ctx context.Context, req Request) (*Result, error) { + if req.FontSize == 0 { + req.FontSize = DefaultFontSize + } + if err := req.validate(); err != nil { + return nil, err + } + + c := &crawler.Crawler{ + Client: crawler.NewClient(DefaultDelay, retries), + CacheDir: req.CacheDir, + Workers: workers, + Log: req.Log, + } + + novel, err := c.Novel(ctx, req.NovelURL) + if err != nil { + return nil, err + } + + chapters, err := c.Chapters(ctx, novel) + if err != nil { + return nil, err + } + + outPath := filepath.Join(req.OutDir, SafeFileName(novel.Title, novel.Slug)+".pdf") + opts := pdf.Options{ + Page: pdf.PhonePage, + Margin: margin, + Font: pdf.BundledFont(), + FontSize: req.FontSize, + LineSpacing: lineSpacing, + Title: novel.Title, + SourceURL: novel.URL, + } + if err := pdf.Write(outPath, opts, toPDFChapters(chapters)); err != nil { + return nil, err + } + + return &Result{ + Path: outPath, + Title: novel.Title, + SourceURL: novel.URL, + Chapters: len(chapters), + Words: crawler.TotalWords(chapters), + Page: pdf.PhonePage, + }, nil +} + +// validate rejects a Request before any request is made, so a typo costs no +// fetches. +func (r *Request) validate() error { + if r.NovelURL == "" { + return fmt.Errorf("novel url is required") + } + parsed, err := url.Parse(r.NovelURL) + if err != nil { + return fmt.Errorf("invalid novel url: %w", err) + } + if parsed.Scheme != "http" && parsed.Scheme != "https" { + return fmt.Errorf("invalid novel url: want an http(s) URL, got %q", r.NovelURL) + } + if r.FontSize <= 0 { + return fmt.Errorf("font size must be positive") + } + return nil +} + +func toPDFChapters(chapters []*crawler.Chapter) []pdf.Chapter { + out := make([]pdf.Chapter, 0, len(chapters)) + for _, ch := range chapters { + out = append(out, pdf.Chapter{Heading: ch.Heading(), Paragraphs: ch.Paragraphs}) + } + return out +} + +var unsafeNameChars = regexp.MustCompile(`[^\p{L}\p{N}]+`) + +// SafeFileName builds a file name from the novel title, falling back to the +// slug when the title has no usable characters. The result has no extension. +func SafeFileName(title, fallback string) string { + name := strings.Trim(unsafeNameChars.ReplaceAllString(title, "-"), "-") + if name == "" { + return fallback + } + return name +} diff --git a/internal/modules/monkeyd/export/export_test.go b/internal/modules/monkeyd/export/export_test.go new file mode 100644 index 0000000..2a9aed1 --- /dev/null +++ b/internal/modules/monkeyd/export/export_test.go @@ -0,0 +1,81 @@ +package export + +import "testing" + +func TestValidate(t *testing.T) { + tests := []struct { + name string + req Request + wantErr bool + }{ + { + name: "url and font size are valid", + req: Request{NovelURL: "https://monkeydd.com/example.html", FontSize: DefaultFontSize}, + }, + { + name: "missing url", + req: Request{FontSize: DefaultFontSize}, + wantErr: true, + }, + { + name: "non-http scheme", + req: Request{NovelURL: "ftp://monkeydd.com/example.html", FontSize: DefaultFontSize}, + wantErr: true, + }, + { + name: "scheme-less url", + req: Request{NovelURL: "monkeydd.com/example.html", FontSize: DefaultFontSize}, + wantErr: true, + }, + { + name: "negative font size", + req: Request{NovelURL: "https://monkeydd.com/example.html", FontSize: -1}, + wantErr: true, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + err := tt.req.validate() + if tt.wantErr && err == nil { + t.Error("validate() = nil, want error") + } + if !tt.wantErr && err != nil { + t.Errorf("validate() = %v, want nil", err) + } + }) + } +} + +// An invalid request must fail before the crawler makes any request. +func TestExportRejectsInvalidRequestWithoutFetching(t *testing.T) { + if _, err := Export(t.Context(), Request{NovelURL: "ftp://monkeydd.com/x.html", OutDir: t.TempDir()}); err == nil { + t.Fatal("Export = nil error, want an invalid url error") + } +} + +func TestSafeFileName(t *testing.T) { + tests := []struct { + title string + fallback string + want string + }{ + {"Trở Lại Năm Tháng Cũ", "slug", "Trở-Lại-Năm-Tháng-Cũ"}, + {"Chapter: One / Two", "slug", "Chapter-One-Two"}, + {" --- ", "slug", "slug"}, + {"", "slug", "slug"}, + } + for _, tt := range tests { + if got := SafeFileName(tt.title, tt.fallback); got != tt.want { + t.Errorf("SafeFileName(%q, %q) = %q, want %q", tt.title, tt.fallback, got, tt.want) + } + } +} + +func TestResultSummary(t *testing.T) { + r := &Result{Title: "Example", Chapters: 12, Words: 3400} + want := "Example — 12 chapters, 3400 words" + if got := r.Summary(); got != want { + t.Errorf("Summary() = %q, want %q", got, want) + } +} diff --git a/internal/modules/monkeyd/export_job.go b/internal/modules/monkeyd/export_job.go index 2091928..7542b09 100644 --- a/internal/modules/monkeyd/export_job.go +++ b/internal/modules/monkeyd/export_job.go @@ -11,7 +11,7 @@ import ( "github.com/go-telegram/bot" "github.com/go-telegram/bot/models" - "github.com/tiennm99/monkeyd-crawler/export" + "github.com/tiennm99/miti99bot/internal/modules/monkeyd/export" "github.com/tiennm99/miti99bot/internal/log" "github.com/tiennm99/miti99bot/internal/modules/util/chathelper" diff --git a/internal/modules/monkeyd/handlers_test.go b/internal/modules/monkeyd/handlers_test.go index 86dd504..e037d4c 100644 --- a/internal/modules/monkeyd/handlers_test.go +++ b/internal/modules/monkeyd/handlers_test.go @@ -9,8 +9,8 @@ import ( "strings" "testing" - "github.com/tiennm99/monkeyd-crawler/export" - "github.com/tiennm99/monkeyd-crawler/pdfout" + "github.com/tiennm99/miti99bot/internal/modules/monkeyd/export" + "github.com/tiennm99/miti99bot/internal/modules/monkeyd/pdf" "github.com/tiennm99/miti99bot/internal/modules" "github.com/tiennm99/miti99bot/internal/storage" @@ -77,7 +77,7 @@ func stubPDF(t *testing.T, dir, name string, size int64) *export.Result { SourceURL: testNovelURL, Chapters: 3, Words: 1200, - Page: pdfout.Presets["phone"], + Page: pdf.PhonePage, } } diff --git a/internal/modules/monkeyd/monkeyd.go b/internal/modules/monkeyd/monkeyd.go index ef7bb40..511877b 100644 --- a/internal/modules/monkeyd/monkeyd.go +++ b/internal/modules/monkeyd/monkeyd.go @@ -15,7 +15,7 @@ import ( "github.com/go-telegram/bot" "github.com/go-telegram/bot/models" - "github.com/tiennm99/monkeyd-crawler/export" + "github.com/tiennm99/miti99bot/internal/modules/monkeyd/export" "github.com/tiennm99/miti99bot/internal/modules" "github.com/tiennm99/miti99bot/internal/modules/util/chathelper" diff --git a/internal/modules/monkeyd/pdf/font.go b/internal/modules/monkeyd/pdf/font.go new file mode 100644 index 0000000..945b459 --- /dev/null +++ b/internal/modules/monkeyd/pdf/font.go @@ -0,0 +1,37 @@ +package pdf + +import ( + _ "embed" +) + +// bundledTTF is the only font the bot renders with. A minimal container has no +// system fonts, and a single embedded font keeps the PDF identical on every +// host. See fonts/NOTICE.md for provenance and licensing. +// +// Vietnamese text needs the Latin Extended Additional block (ư, ạ, ế, ộ …); +// DejaVu Sans covers it, where a basic-Latin font would silently drop the +// diacritics. +// +//go:embed fonts/DejaVuSans.ttf +var bundledTTF []byte + +// BundledFontName labels the embedded font in diagnostics. It is not a path; +// the font is compiled into the binary. +const BundledFontName = "DejaVu Sans (bundled)" + +// Font is font data ready to embed in a PDF. +// +// The data is carried as bytes rather than as a path because fpdf joins a font +// path onto its own font directory, which it defaults to "." — turning an +// absolute path into a working-directory-relative one that resolves only when +// the process happens to run from the filesystem root. +type Font struct { + // Name identifies the font for diagnostics. + Name string + Data []byte +} + +// BundledFont returns the font compiled into the binary. +func BundledFont() Font { + return Font{Name: BundledFontName, Data: bundledTTF} +} diff --git a/internal/modules/monkeyd/pdf/font_test.go b/internal/modules/monkeyd/pdf/font_test.go new file mode 100644 index 0000000..973d92b --- /dev/null +++ b/internal/modules/monkeyd/pdf/font_test.go @@ -0,0 +1,38 @@ +package pdf + +import ( + "testing" + + "golang.org/x/image/font/sfnt" +) + +func TestBundledFontIsParseable(t *testing.T) { + font := BundledFont() + if font.Name != BundledFontName { + t.Errorf("Name = %q, want %q", font.Name, BundledFontName) + } + if _, err := sfnt.Parse(font.Data); err != nil { + t.Fatalf("bundled font does not parse as a TrueType font: %v", err) + } +} + +// The bundled font exists to render Vietnamese. A replacement that lacks these +// glyphs would silently emit blanks, so check before trusting it. +func TestBundledFontCoversVietnamese(t *testing.T) { + parsed, err := sfnt.Parse(BundledFont().Data) + if err != nil { + t.Fatalf("parse bundled font: %v", err) + } + var buf sfnt.Buffer + for _, r := range []rune{'ư', 'ơ', 'đ', 'ạ', 'ế', 'ộ', 'ữ', 'ằ', 'ỷ', 'ỹ', 'Ọ', 'Ế', 'Ư', 'Đ'} { + index, err := parsed.GlyphIndex(&buf, r) + if err != nil { + t.Errorf("GlyphIndex(%q): %v", r, err) + continue + } + // Glyph 0 is .notdef — the character is absent from the font. + if index == 0 { + t.Errorf("bundled font has no glyph for %q (U+%04X)", r, r) + } + } +} diff --git a/internal/modules/monkeyd/pdf/fonts/DejaVuSans.ttf b/internal/modules/monkeyd/pdf/fonts/DejaVuSans.ttf new file mode 100644 index 0000000..e5f7eec Binary files /dev/null and b/internal/modules/monkeyd/pdf/fonts/DejaVuSans.ttf differ diff --git a/internal/modules/monkeyd/pdf/fonts/NOTICE.md b/internal/modules/monkeyd/pdf/fonts/NOTICE.md new file mode 100644 index 0000000..9394353 --- /dev/null +++ b/internal/modules/monkeyd/pdf/fonts/NOTICE.md @@ -0,0 +1,24 @@ +# Bundled font + +`DejaVuSans.ttf` is embedded into the binary and used when no font is supplied +and no suitable system font is found. It covers the Latin Extended Additional +block, which is what Vietnamese diacritics need — a font with only basic Latin +coverage silently drops them. + +The following is recorded in the font file's own name table: + +- Version: `Version 2.37` +- Copyright: `Copyright (c) 2003 by Bitstream, Inc. All Rights Reserved.` + `Copyright (c) 2006 by Tavmjong Bah. All Rights Reserved.` + `DejaVu changes are in public domain` +- License information: + +The DejaVu fonts are free and redistributable, which is why they ship with most +Linux distributions. This copy came from Alpine's `font-dejavu` package, which +does not include the license text as a separate file. For a vendored copy of the +full license text, take `LICENSE` from the upstream DejaVu release and add it to +this directory. + +To swap the bundled font, replace `DejaVuSans.ttf` and update this file. Verify +the replacement covers Vietnamese first — `TestBundledFontCoversVietnamese` in +`../font_test.go` checks a representative set of characters. diff --git a/internal/modules/monkeyd/pdf/pdf.go b/internal/modules/monkeyd/pdf/pdf.go new file mode 100644 index 0000000..fc47a42 --- /dev/null +++ b/internal/modules/monkeyd/pdf/pdf.go @@ -0,0 +1,132 @@ +package pdf + +import ( + "fmt" + "strings" + + "github.com/go-pdf/fpdf" +) + +// pointsToMM converts typographic points to millimetres. +const pointsToMM = 25.4 / 72.0 + +// bodyFont is the internal family name registered with the PDF. +const bodyFont = "body" + +// PageSize is a page in millimetres. +type PageSize struct { + Name string + W, H float64 +} + +// PhonePage is the page every export uses. +// +// Phone reading depends far more on page shape than on font size. A phone +// viewer scales a whole page to fit the screen, so a large font on an A4 page +// still ends up tiny: the page is ~3x wider than the screen and gets shrunk to +// match. A page cut to the phone's own aspect ratio fills the screen at 100%, +// which is why this is a small 9:16 page rather than A4 with big type. +var PhonePage = PageSize{Name: "phone", W: 90, H: 160} + +// Chapter is a chapter ready to render. +type Chapter struct { + Heading string + Paragraphs []string +} + +// Options controls the exported PDF. +type Options struct { + Page PageSize + Margin float64 // mm + Font Font // BundledFont() + FontSize float64 // pt + LineSpacing float64 // multiple of font size + Title string + SourceURL string +} + +// footerReserve is the vertical space kept clear for the page number. +const footerReserve = 6.0 + +// Write renders the chapters to a PDF at path. +func Write(path string, opts Options, chapters []Chapter) error { + pdf := fpdf.NewCustom(&fpdf.InitType{ + UnitStr: "mm", + Size: fpdf.SizeType{Wd: opts.Page.W, Ht: opts.Page.H}, + }) + + pdf.SetMargins(opts.Margin, opts.Margin, opts.Margin) + pdf.SetAutoPageBreak(true, opts.Margin+footerReserve) + + // Embeds a subset of the TrueType data, which is what makes the Vietnamese + // diacritics render instead of falling back to "?". The bytes are passed + // directly rather than by path: the path-taking variant joins the name onto + // fpdf's own font directory (default "."), which mangles an absolute path + // into a working-directory-relative one. + pdf.AddUTF8FontFromBytes(bodyFont, "", opts.Font.Data) + pdf.SetFont(bodyFont, "", opts.FontSize) + pdf.SetTitle(opts.Title, true) + + lineHeight := opts.FontSize * opts.LineSpacing * pointsToMM + paragraphGap := lineHeight * 0.45 + headingSize := opts.FontSize * 1.35 + + addFooter(pdf, opts) + writeTitlePage(pdf, opts, len(chapters)) + + for _, ch := range chapters { + pdf.AddPage() + + pdf.SetFontSize(headingSize) + pdf.MultiCell(0, headingSize*1.3*pointsToMM, ch.Heading, "", "L", false) + pdf.Ln(paragraphGap * 1.6) + + pdf.SetFontSize(opts.FontSize) + for _, p := range ch.Paragraphs { + // "J" justifies, which keeps the short measure of a phone page tidy. + pdf.MultiCell(0, lineHeight, p, "", "J", false) + pdf.Ln(paragraphGap) + } + } + + if err := pdf.OutputFileAndClose(path); err != nil { + return fmt.Errorf("write pdf %s: %w", path, err) + } + return nil +} + +// addFooter prints a centred page number, restoring the body font size so the +// footer callback cannot leak its own size into the following content. +func addFooter(pdf *fpdf.Fpdf, opts Options) { + pdf.SetFooterFunc(func() { + if pdf.PageNo() <= 1 { + return + } + pdf.SetY(-(opts.Margin + footerReserve*0.6)) + pdf.SetFontSize(opts.FontSize * 0.75) + pdf.SetTextColor(120, 120, 120) + pdf.CellFormat(0, 4, fmt.Sprintf("%d", pdf.PageNo()-1), "", 0, "C", false, 0, "") + pdf.SetTextColor(0, 0, 0) + pdf.SetFontSize(opts.FontSize) + }) +} + +func writeTitlePage(pdf *fpdf.Fpdf, opts Options, chapterCount int) { + pdf.AddPage() + pdf.SetY(opts.Page.H * 0.30) + + titleSize := opts.FontSize * 1.9 + pdf.SetFontSize(titleSize) + pdf.MultiCell(0, titleSize*1.35*pointsToMM, strings.ToUpper(opts.Title), "", "C", false) + + pdf.Ln(opts.FontSize * pointsToMM * 2) + pdf.SetFontSize(opts.FontSize * 0.85) + pdf.SetTextColor(90, 90, 90) + pdf.MultiCell(0, opts.FontSize*1.4*pointsToMM, + fmt.Sprintf("%d chương", chapterCount), "", "C", false) + if opts.SourceURL != "" { + pdf.MultiCell(0, opts.FontSize*1.4*pointsToMM, opts.SourceURL, "", "C", false) + } + pdf.SetTextColor(0, 0, 0) + pdf.SetFontSize(opts.FontSize) +} diff --git a/internal/modules/monkeyd/pdf/pdf_test.go b/internal/modules/monkeyd/pdf/pdf_test.go new file mode 100644 index 0000000..392c5ab --- /dev/null +++ b/internal/modules/monkeyd/pdf/pdf_test.go @@ -0,0 +1,79 @@ +package pdf + +import ( + "os" + "path/filepath" + "testing" +) + +func testOptions(t *testing.T) Options { + t.Helper() + return Options{ + Page: PhonePage, + Margin: 6, + Font: BundledFont(), + FontSize: 12, + LineSpacing: 1.55, + Title: "TRỞ LẠI NĂM THÁNG CŨ", + SourceURL: "https://example.test/n.html", + } +} + +func TestWriteProducesReadablePDF(t *testing.T) { + path := filepath.Join(t.TempDir(), "out.pdf") + + chapters := []Chapter{ + {Heading: "Chương 1", Paragraphs: []string{ + "Nghe vị trưởng tử nói chuyện với nàng.", + "Một đoạn văn khác để kiểm tra ngắt dòng tự động trên trang nhỏ.", + }}, + {Heading: "Chương 2", Paragraphs: []string{"Đoạn cuối."}}, + } + + if err := Write(path, testOptions(t), chapters); err != nil { + t.Fatalf("Write: %v", err) + } + + info, err := os.Stat(path) + if err != nil { + t.Fatalf("stat output: %v", err) + } + if info.Size() == 0 { + t.Fatal("wrote an empty PDF") + } + + header := make([]byte, 5) + f, err := os.Open(path) + if err != nil { + t.Fatal(err) + } + defer f.Close() + if _, err := f.Read(header); err != nil { + t.Fatal(err) + } + if string(header) != "%PDF-" { + t.Errorf("output does not start with a PDF header, got %q", header) + } +} + +// A novel with many chapters must not overflow a page; auto page break plus the +// footer reserve handles that, so a long chapter should span several pages. +func TestWriteHandlesLongChapters(t *testing.T) { + path := filepath.Join(t.TempDir(), "long.pdf") + + paragraphs := make([]string, 200) + for i := range paragraphs { + paragraphs[i] = "Một đoạn văn dài để buộc trình kết xuất phải sang trang mới nhiều lần." + } + + if err := Write(path, testOptions(t), []Chapter{{Heading: "Chương 1", Paragraphs: paragraphs}}); err != nil { + t.Fatalf("Write: %v", err) + } + info, err := os.Stat(path) + if err != nil { + t.Fatal(err) + } + if info.Size() < 2000 { + t.Errorf("output suspiciously small (%d bytes) for 200 paragraphs", info.Size()) + } +} diff --git a/internal/modules/monkeyd/tags_command.go b/internal/modules/monkeyd/tags_command.go index b800441..9d225f5 100644 --- a/internal/modules/monkeyd/tags_command.go +++ b/internal/modules/monkeyd/tags_command.go @@ -12,8 +12,8 @@ import ( "github.com/go-telegram/bot" "github.com/go-telegram/bot/models" - "github.com/tiennm99/monkeyd-crawler/export" - crawler "github.com/tiennm99/monkeyd-crawler/monkeyd" + "github.com/tiennm99/miti99bot/internal/modules/monkeyd/crawler" + "github.com/tiennm99/miti99bot/internal/modules/monkeyd/export" "github.com/tiennm99/miti99bot/internal/log" "github.com/tiennm99/miti99bot/internal/modules" diff --git a/third_party/monkeyd-crawler b/third_party/monkeyd-crawler deleted file mode 160000 index d88f2a4..0000000 --- a/third_party/monkeyd-crawler +++ /dev/null @@ -1 +0,0 @@ -Subproject commit d88f2a482d84322aae0bde9b98a1d6985ef36b17