mirror of
https://github.com/tiennm99/tiennm99bot.git
synced 2026-10-11 03:13:46 +00:00
refactor(monkeyd): port the crawler in-tree and drop the submodule
The monkeyd-crawler repository moved into tiennm99/mttools, so recursive clones of the old submodule URL fail and block every deploy. The crawler, PDF renderer, and export flow now live under internal/modules/monkeyd, trimmed to the bot's use: a fixed phone page, the bundled font only, and the font size as the single export option.
This commit is contained in:
1 parent
c1393f0ec8
commit
d211124aae
26 files changed
+1818
-30
No files matched your search
@@ -16,12 +16,7 @@ jobs:
|
||||
matrix:
|
||||
go: ['1.26.5']
|
||||
steps:
|
||||
# The monkeyd module builds against third_party/monkeyd-crawler, which is
|
||||
# a submodule wired in through a go.mod replace directive. Without it
|
||||
# checked out, every Go step fails to resolve the package.
|
||||
- uses: actions/checkout@v6
|
||||
with:
|
||||
submodules: true
|
||||
|
||||
- uses: actions/setup-go@v6
|
||||
with:
|
||||
|
||||
@@ -1,3 +0,0 @@
|
||||
[submodule "third_party/monkeyd-crawler"]
|
||||
path = third_party/monkeyd-crawler
|
||||
url = https://github.com/tiennm99/monkeyd-crawler.git
|
||||
@@ -50,6 +50,9 @@ linters:
|
||||
- name: unused-parameter
|
||||
disabled: true
|
||||
exclusions:
|
||||
# renderer/ is the Node service; its node_modules can carry stray Go files.
|
||||
paths:
|
||||
- renderer
|
||||
rules:
|
||||
# Tests routinely pass dummy values, swallow errors from helpers, and
|
||||
# use higher cyclomatic complexity in table-driven cases. Suppress the
|
||||
|
||||
+2
-5
@@ -1,12 +1,9 @@
|
||||
FROM golang:1.26.5-alpine AS builder
|
||||
WORKDIR /src
|
||||
|
||||
# The monkeyd-crawler submodule is resolved through a `replace` directive, so
|
||||
# its go.mod must be present before `go mod download` can read the build list.
|
||||
# Only the module files are copied here, keeping this layer cached across
|
||||
# ordinary source edits.
|
||||
# Only the module files are copied first, keeping the download layer cached
|
||||
# across ordinary source edits.
|
||||
COPY go.mod go.sum ./
|
||||
COPY third_party/monkeyd-crawler/go.mod third_party/monkeyd-crawler/go.sum ./third_party/monkeyd-crawler/
|
||||
RUN go mod download
|
||||
|
||||
COPY . .
|
||||
|
||||
@@ -3,22 +3,19 @@ module github.com/tiennm99/miti99bot
|
||||
go 1.26.5
|
||||
|
||||
require (
|
||||
github.com/go-pdf/fpdf v0.9.0
|
||||
github.com/go-telegram/bot v1.20.0
|
||||
github.com/ledongthuc/pdf v0.0.0-20260907135840-6c8c28e0e8a0
|
||||
github.com/robfig/cron/v3 v3.0.1
|
||||
github.com/testcontainers/testcontainers-go v0.43.0
|
||||
github.com/testcontainers/testcontainers-go/modules/mongodb v0.43.0
|
||||
github.com/tiennm99/monkeyd-crawler v0.0.0
|
||||
go.mongodb.org/mongo-driver/v2 v2.7.0
|
||||
golang.org/x/image v0.45.0
|
||||
golang.org/x/net v0.57.0
|
||||
golang.org/x/sync v0.22.0
|
||||
golang.org/x/text v0.41.0
|
||||
)
|
||||
|
||||
require (
|
||||
github.com/go-pdf/fpdf v0.9.0 // indirect
|
||||
golang.org/x/net v0.57.0 // indirect
|
||||
)
|
||||
|
||||
require (
|
||||
dario.cat/mergo v1.0.2 // indirect
|
||||
github.com/Azure/go-ansiterm v0.0.0-20250102033503-faa5f7b0171c // indirect
|
||||
@@ -72,9 +69,6 @@ require (
|
||||
go.opentelemetry.io/otel/metric v1.41.0 // indirect
|
||||
go.opentelemetry.io/otel/trace v1.41.0 // indirect
|
||||
golang.org/x/crypto v0.54.0 // indirect
|
||||
golang.org/x/sync v0.22.0 // indirect
|
||||
golang.org/x/sys v0.47.0 // indirect
|
||||
gopkg.in/yaml.v3 v3.0.1 // indirect
|
||||
)
|
||||
|
||||
replace github.com/tiennm99/monkeyd-crawler => ./third_party/monkeyd-crawler
|
||||
@@ -0,0 +1,187 @@
|
||||
package crawler
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"fmt"
|
||||
"strconv"
|
||||
"strings"
|
||||
|
||||
"golang.org/x/net/html"
|
||||
)
|
||||
|
||||
// contentElementID is the container holding a chapter's body text.
|
||||
const contentElementID = "chapter-content-render"
|
||||
|
||||
// Chapter is one fetched chapter, reduced to plain paragraphs.
|
||||
type Chapter struct {
|
||||
Label string
|
||||
URL string
|
||||
Paragraphs []string
|
||||
}
|
||||
|
||||
// Heading is the chapter title to print. Labels are often a bare number, which
|
||||
// reads poorly as a heading, so those get the Vietnamese word for "chapter".
|
||||
func (c *Chapter) Heading() string {
|
||||
label := strings.TrimSpace(c.Label)
|
||||
if label == "" {
|
||||
return "Chương"
|
||||
}
|
||||
if _, err := strconv.Atoi(label); err == nil {
|
||||
return "Chương " + label
|
||||
}
|
||||
return label
|
||||
}
|
||||
|
||||
// WordCount is a rough word count, used to sanity-check extraction.
|
||||
func (c *Chapter) WordCount() int {
|
||||
n := 0
|
||||
for _, p := range c.Paragraphs {
|
||||
n += len(strings.Fields(p))
|
||||
}
|
||||
return n
|
||||
}
|
||||
|
||||
// blockTags end the current paragraph when opened or closed.
|
||||
var blockTags = map[string]bool{
|
||||
"p": true, "div": true, "br": true, "hr": true, "blockquote": true,
|
||||
"h1": true, "h2": true, "h3": true, "h4": true, "h5": true, "h6": true,
|
||||
"li": true, "ul": true, "ol": true, "tr": true,
|
||||
}
|
||||
|
||||
// skipTags never contribute prose.
|
||||
//
|
||||
// Anchors are included because inside a chapter body they are always site
|
||||
// chrome: the prev/next chapter navigation and the sponsor call-to-action are
|
||||
// both links, while novel prose never needs one. Inline emphasis tags (b, i,
|
||||
// em) are deliberately absent so italics in the prose survive; the site's icons
|
||||
// use <i> but carry no text.
|
||||
var skipTags = map[string]bool{
|
||||
"script": true, "style": true, "noscript": true, "iframe": true,
|
||||
"ins": true, "form": true, "select": true, "button": true, "textarea": true,
|
||||
"a": true, "img": true, "svg": true,
|
||||
}
|
||||
|
||||
// junkClasses marks containers the site injects into the chapter body. Their
|
||||
// whole subtree is dropped.
|
||||
//
|
||||
// These are matched on class rather than position because the blocks move: the
|
||||
// sponsor block opens the body on most chapters but is absent on others, and
|
||||
// the watermark is planted at a different paragraph in every chapter. Only
|
||||
// site-specific class names are listed; generic Bootstrap utilities such as
|
||||
// "my-4" or "text-center" are not, since prose could legitimately carry them.
|
||||
//
|
||||
// Note the sibling class "actac" is NOT junk: it wraps the real chapter text and
|
||||
// carries style="display:none", because the site gates the body behind a click
|
||||
// on the sponsor link and reveals it with JavaScript. Skipping hidden elements,
|
||||
// or skipping "act*" as a family, would therefore discard the whole chapter.
|
||||
var junkClasses = map[string]bool{
|
||||
"actcl": true, // sponsor block shown in place of the gated chapter body
|
||||
"signature": true, // "[Truyện được đăng tải duy nhất tại ...]" source watermark
|
||||
}
|
||||
|
||||
// hasJunkClass reports whether a node is an injected non-prose container.
|
||||
func hasJunkClass(n *html.Node) bool {
|
||||
for _, tok := range strings.Fields(attr(n, "class")) {
|
||||
if junkClasses[tok] {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// ParseChapter extracts a chapter's paragraphs, restoring the words the site
|
||||
// serves through CSS :before rules instead of markup.
|
||||
func ParseChapter(page []byte, ref ChapterRef) (*Chapter, error) {
|
||||
doc, err := html.Parse(bytes.NewReader(page))
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("parse chapter %s: %w", ref.URL, err)
|
||||
}
|
||||
content := elementByID(doc, contentElementID)
|
||||
if content == nil {
|
||||
return nil, fmt.Errorf("chapter %s: no #%s container (page layout may have changed)",
|
||||
ref.URL, contentElementID)
|
||||
}
|
||||
|
||||
ch := &Chapter{
|
||||
Label: ref.Label,
|
||||
URL: ref.URL,
|
||||
Paragraphs: extractParagraphs(content, ParseWordClasses(page)),
|
||||
}
|
||||
if len(ch.Paragraphs) == 0 {
|
||||
return nil, fmt.Errorf("chapter %s: extracted no text", ref.URL)
|
||||
}
|
||||
ch.Paragraphs = dropRepeatedTitle(ch.Paragraphs, ref.Label)
|
||||
return ch, nil
|
||||
}
|
||||
|
||||
// extractParagraphs walks the content subtree into plain paragraphs, replacing
|
||||
// each word-carrying element with the word its CSS rule injects.
|
||||
func extractParagraphs(content *html.Node, words map[string]string) []string {
|
||||
var b strings.Builder
|
||||
|
||||
var walk func(*html.Node)
|
||||
walk = func(n *html.Node) {
|
||||
switch n.Type {
|
||||
case html.TextNode:
|
||||
// The HTML parser has already decoded entities such as ư.
|
||||
b.WriteString(n.Data)
|
||||
return
|
||||
case html.ElementNode:
|
||||
if skipTags[n.Data] || hasJunkClass(n) {
|
||||
return
|
||||
}
|
||||
// These elements are empty in the markup; the CSS word replaces them.
|
||||
if word, ok := injectedWord(n, words); ok {
|
||||
b.WriteString(word)
|
||||
return
|
||||
}
|
||||
if blockTags[n.Data] {
|
||||
b.WriteByte('\n')
|
||||
}
|
||||
}
|
||||
|
||||
for c := n.FirstChild; c != nil; c = c.NextSibling {
|
||||
walk(c)
|
||||
}
|
||||
|
||||
if n.Type == html.ElementNode && blockTags[n.Data] {
|
||||
b.WriteByte('\n')
|
||||
}
|
||||
}
|
||||
walk(content)
|
||||
|
||||
var paragraphs []string
|
||||
for _, line := range strings.Split(b.String(), "\n") {
|
||||
if p := collapseSpaces(line); p != "" {
|
||||
paragraphs = append(paragraphs, p)
|
||||
}
|
||||
}
|
||||
return paragraphs
|
||||
}
|
||||
|
||||
// injectedWord returns the word a node's class supplies via CSS, if any.
|
||||
func injectedWord(n *html.Node, words map[string]string) (string, bool) {
|
||||
class := attr(n, "class")
|
||||
if class == "" {
|
||||
return "", false
|
||||
}
|
||||
for _, tok := range strings.Fields(class) {
|
||||
if word, ok := words[tok]; ok {
|
||||
return word, true
|
||||
}
|
||||
}
|
||||
return "", false
|
||||
}
|
||||
|
||||
// dropRepeatedTitle removes a leading paragraph that only repeats the chapter
|
||||
// label, since the export prints its own heading.
|
||||
func dropRepeatedTitle(paragraphs []string, label string) []string {
|
||||
if len(paragraphs) < 2 {
|
||||
return paragraphs
|
||||
}
|
||||
first := strings.TrimSpace(paragraphs[0])
|
||||
if strings.EqualFold(first, strings.TrimSpace(label)) {
|
||||
return paragraphs[1:]
|
||||
}
|
||||
return paragraphs
|
||||
}
|
||||
@@ -0,0 +1,200 @@
|
||||
package crawler
|
||||
|
||||
import (
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// chapterFixture mirrors the real page shape: words split between markup and
|
||||
// CSS :before rules, HTML entities, spacer paragraphs and an ad script
|
||||
// inside the content container.
|
||||
const chapterFixture = `<!DOCTYPE html><html><head>
|
||||
<style>
|
||||
.t-aaa:before { content: "v\1ecb "; }
|
||||
.j-bbb:before { content: "tr\1ed3 "; }
|
||||
.z-ccc::before{content:"n\E0ng";}
|
||||
.unused-ddd:before { content: "khong-dung"; }
|
||||
</style></head><body>
|
||||
<h1 class="card-title">TEN TRUYEN - 1</h1>
|
||||
<div class="content-container" id="chapter-content-render">
|
||||
<p>1</p>
|
||||
<p> </p>
|
||||
<p>Nghe <span class="t-aaa"></span> trưởng tử noi.</p>
|
||||
<p> </p>
|
||||
<p>Một <span class="j-bbb"></span> và <span class="z-ccc"></span> di.</p>
|
||||
<script>ads();</script>
|
||||
</div></body></html>`
|
||||
|
||||
func TestParseWordClassesDecodesEscapes(t *testing.T) {
|
||||
words := ParseWordClasses([]byte(chapterFixture))
|
||||
|
||||
for class, want := range map[string]string{
|
||||
"t-aaa": "vị",
|
||||
"j-bbb": "trồ",
|
||||
"z-ccc": "nàng",
|
||||
} {
|
||||
if got := words[class]; got != want {
|
||||
t.Errorf("words[%q] = %q, want %q", class, got, want)
|
||||
}
|
||||
}
|
||||
if len(words) != 4 {
|
||||
t.Errorf("got %d rules, want 4", len(words))
|
||||
}
|
||||
}
|
||||
|
||||
func TestParseChapterRestoresCSSWords(t *testing.T) {
|
||||
ch, err := ParseChapter([]byte(chapterFixture), ChapterRef{Label: "1", URL: "http://x/1.html"})
|
||||
if err != nil {
|
||||
t.Fatalf("ParseChapter: %v", err)
|
||||
}
|
||||
|
||||
want := []string{
|
||||
"Nghe vị trưởng tử noi.",
|
||||
"Một trồ và nàng di.",
|
||||
}
|
||||
if len(ch.Paragraphs) != len(want) {
|
||||
t.Fatalf("got %d paragraphs %q, want %d", len(ch.Paragraphs), ch.Paragraphs, len(want))
|
||||
}
|
||||
for i, w := range want {
|
||||
if ch.Paragraphs[i] != w {
|
||||
t.Errorf("paragraph %d = %q, want %q", i, ch.Paragraphs[i], w)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// The CSS-injected words are the difference between real text and text with
|
||||
// silent holes, so guard against a regression that drops them.
|
||||
func TestParseChapterWithoutCSSWouldLoseWords(t *testing.T) {
|
||||
withoutCSS := strings.Replace(chapterFixture, `.t-aaa:before { content: "v\1ecb "; }`, "", 1)
|
||||
|
||||
ch, err := ParseChapter([]byte(withoutCSS), ChapterRef{Label: "1", URL: "http://x/1.html"})
|
||||
if err != nil {
|
||||
t.Fatalf("ParseChapter: %v", err)
|
||||
}
|
||||
if strings.Contains(ch.Paragraphs[0], "vị") {
|
||||
t.Fatal("word appeared without its CSS rule; fixture no longer proves anything")
|
||||
}
|
||||
if want := "Nghe trưởng tử noi."; ch.Paragraphs[0] != want {
|
||||
t.Errorf("paragraph 0 = %q, want %q", ch.Paragraphs[0], want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestParseChapterDropsScriptsAndSpacers(t *testing.T) {
|
||||
ch, err := ParseChapter([]byte(chapterFixture), ChapterRef{Label: "1", URL: "http://x/1.html"})
|
||||
if err != nil {
|
||||
t.Fatalf("ParseChapter: %v", err)
|
||||
}
|
||||
for _, p := range ch.Paragraphs {
|
||||
if strings.Contains(p, "ads()") {
|
||||
t.Errorf("script text leaked into paragraph %q", p)
|
||||
}
|
||||
if strings.TrimSpace(p) == "" {
|
||||
t.Error("empty spacer paragraph was kept")
|
||||
}
|
||||
if strings.Contains(p, " ") {
|
||||
t.Errorf("non-breaking space survived in %q", p)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// The leading "<p>1</p>" repeats the chapter label and would print twice.
|
||||
func TestParseChapterDropsRepeatedTitle(t *testing.T) {
|
||||
ch, err := ParseChapter([]byte(chapterFixture), ChapterRef{Label: "1", URL: "http://x/1.html"})
|
||||
if err != nil {
|
||||
t.Fatalf("ParseChapter: %v", err)
|
||||
}
|
||||
if ch.Paragraphs[0] == "1" {
|
||||
t.Error("repeated chapter label was kept as a paragraph")
|
||||
}
|
||||
}
|
||||
|
||||
// gatedChapterFixture mirrors how most chapters are served: a visible sponsor
|
||||
// block (div.actcl) stands in for the body, while the real text sits in a
|
||||
// sibling div.actac hidden with display:none and revealed by the site's
|
||||
// JavaScript. A source watermark and prev/next navigation bracket the prose.
|
||||
const gatedChapterFixture = `<!DOCTYPE html><html><head>
|
||||
<style>.t-aaa:before { content: "v\1ecb"; }</style></head><body>
|
||||
<div class="content-container" id="chapter-content-render">
|
||||
<div class="actcl">
|
||||
<h4 class="text-center text-primary">Moi Quy doc gia CLICK vao lien ket</h4>
|
||||
<p class="text-center">mo ung dung Shopee, sau do quay tro lai de doc!</p>
|
||||
<a class="btn btn-primary px-3" href="https://s.shopee.vn/xxxx">
|
||||
<img src="x.jpg"><span class="text-uppercase text-danger">CLICK</span></a>
|
||||
<h4 class="text-center text-primary">MonkeyD va doi ngu Editor xin chan thanh cam on!</h4>
|
||||
</div>
|
||||
<div class="actac" style=" display:none; ">
|
||||
<p>Doan van dau tien co <span class="t-aaa"></span> tri.</p>
|
||||
<p> </p>
|
||||
<p class="signature">[Truyen duoc dang tai duy nhat tai MonkeyDD.com - https://monkeydd.com/n/10.html.]</p>
|
||||
<p>Doan van cuoi cung.</p>
|
||||
</div>
|
||||
<div class="my-4"><div class="d-flex justify-content-center">
|
||||
<a class="btn btn-primary px-3 me-2" href="/n/9.html"><i class="bx bx-chevron-left"></i>Chương trước</a>
|
||||
<a class="btn btn-primary px-3" href="/n/11.html">Chương sau<i class="bx bx-chevron-right"></i></a>
|
||||
</div></div>
|
||||
</div></body></html>`
|
||||
|
||||
// The gated body is the one thing that must survive: it is hidden with
|
||||
// display:none, so any rule that drops hidden or "act*" containers silently
|
||||
// discards the entire chapter.
|
||||
func TestParseChapterKeepsGatedBody(t *testing.T) {
|
||||
ch, err := ParseChapter([]byte(gatedChapterFixture), ChapterRef{Label: "10", URL: "http://x/10.html"})
|
||||
if err != nil {
|
||||
t.Fatalf("ParseChapter: %v", err)
|
||||
}
|
||||
|
||||
want := []string{
|
||||
"Doan van dau tien co vị tri.",
|
||||
"Doan van cuoi cung.",
|
||||
}
|
||||
if len(ch.Paragraphs) != len(want) {
|
||||
t.Fatalf("got %d paragraphs %q, want %d", len(ch.Paragraphs), ch.Paragraphs, len(want))
|
||||
}
|
||||
for i, w := range want {
|
||||
if ch.Paragraphs[i] != w {
|
||||
t.Errorf("paragraph %d = %q, want %q", i, ch.Paragraphs[i], w)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Everything the site injects around the prose must be gone.
|
||||
func TestParseChapterDropsInjectedBlocks(t *testing.T) {
|
||||
ch, err := ParseChapter([]byte(gatedChapterFixture), ChapterRef{Label: "10", URL: "http://x/10.html"})
|
||||
if err != nil {
|
||||
t.Fatalf("ParseChapter: %v", err)
|
||||
}
|
||||
body := strings.Join(ch.Paragraphs, "\n")
|
||||
|
||||
for _, junk := range []string{
|
||||
"Shopee", // sponsor copy
|
||||
"CLICK", // sponsor call to action
|
||||
"cam on", // sponsor sign-off
|
||||
"MonkeyDD.com", // source watermark
|
||||
"Chương trước", // navigation
|
||||
"Chương sau", // navigation
|
||||
} {
|
||||
if strings.Contains(body, junk) {
|
||||
t.Errorf("injected text %q survived extraction in %q", junk, body)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestParseChapterMissingContainer(t *testing.T) {
|
||||
if _, err := ParseChapter([]byte(`<html><body><p>hi</p></body></html>`),
|
||||
ChapterRef{URL: "http://x/1.html"}); err == nil {
|
||||
t.Fatal("want an error when the content container is absent")
|
||||
}
|
||||
}
|
||||
|
||||
func TestChapterHeading(t *testing.T) {
|
||||
for _, tc := range []struct{ label, want string }{
|
||||
{"14", "Chương 14"},
|
||||
{"Chương 12", "Chương 12"},
|
||||
{"", "Chương"},
|
||||
} {
|
||||
ch := &Chapter{Label: tc.label}
|
||||
if got := ch.Heading(); got != tc.want {
|
||||
t.Errorf("Heading(%q) = %q, want %q", tc.label, got, tc.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,135 @@
|
||||
package crawler
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"io"
|
||||
"net/http"
|
||||
"sync"
|
||||
"time"
|
||||
)
|
||||
|
||||
// The site rejects requests without a browser-like User-Agent.
|
||||
const defaultUserAgent = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 " +
|
||||
"(KHTML, like Gecko) Chrome/124.0.0.0 Safari/537.36"
|
||||
|
||||
// maxPageSize caps how much of a response we buffer; chapter pages are ~130 KB.
|
||||
const maxPageSize = 8 << 20
|
||||
|
||||
// statusError reports an unexpected HTTP status so Get can decide whether
|
||||
// retrying is worthwhile.
|
||||
type statusError struct {
|
||||
code int
|
||||
status string
|
||||
}
|
||||
|
||||
func (e *statusError) Error() string { return "unexpected status " + e.status }
|
||||
|
||||
// retryable is true for transient failures. A 404 or 403 will not fix itself,
|
||||
// so those fail immediately instead of burning the retry budget.
|
||||
func (e *statusError) retryable() bool {
|
||||
return e.code == http.StatusTooManyRequests || e.code >= 500
|
||||
}
|
||||
|
||||
// Client fetches pages from monkeydd.com. It spaces requests out by a fixed
|
||||
// delay no matter how many goroutines call Get, so raising the worker count
|
||||
// never raises the request rate, and it retries transient failures with
|
||||
// exponential backoff.
|
||||
type Client struct {
|
||||
http *http.Client
|
||||
ua string
|
||||
delay time.Duration
|
||||
retries int
|
||||
|
||||
mu sync.Mutex
|
||||
nextSlot time.Time
|
||||
}
|
||||
|
||||
func NewClient(delay time.Duration, retries int) *Client {
|
||||
return &Client{
|
||||
http: &http.Client{Timeout: 45 * time.Second},
|
||||
ua: defaultUserAgent,
|
||||
delay: delay,
|
||||
retries: retries,
|
||||
}
|
||||
}
|
||||
|
||||
// reserve claims the next request slot and blocks until it comes due, holding
|
||||
// the global rate at one request per delay across all callers.
|
||||
func (c *Client) reserve(ctx context.Context) error {
|
||||
c.mu.Lock()
|
||||
slot := c.nextSlot
|
||||
if now := time.Now(); slot.Before(now) {
|
||||
slot = now
|
||||
}
|
||||
c.nextSlot = slot.Add(c.delay)
|
||||
c.mu.Unlock()
|
||||
|
||||
wait := time.Until(slot)
|
||||
if wait <= 0 {
|
||||
return nil
|
||||
}
|
||||
timer := time.NewTimer(wait)
|
||||
defer timer.Stop()
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return ctx.Err()
|
||||
case <-timer.C:
|
||||
return nil
|
||||
}
|
||||
}
|
||||
|
||||
// Get fetches url, retrying transient failures.
|
||||
func (c *Client) Get(ctx context.Context, url string) ([]byte, error) {
|
||||
var lastErr error
|
||||
for attempt := 0; attempt <= c.retries; attempt++ {
|
||||
if attempt > 0 {
|
||||
backoff := time.Duration(1<<uint(attempt-1)) * time.Second
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return nil, ctx.Err()
|
||||
case <-time.After(backoff):
|
||||
}
|
||||
}
|
||||
if err := c.reserve(ctx); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
body, err := c.fetch(ctx, url)
|
||||
if err == nil {
|
||||
return body, nil
|
||||
}
|
||||
lastErr = err
|
||||
|
||||
var se *statusError
|
||||
if errors.As(err, &se) && !se.retryable() {
|
||||
break
|
||||
}
|
||||
if ctx.Err() != nil {
|
||||
break
|
||||
}
|
||||
}
|
||||
return nil, fmt.Errorf("fetch %s: %w", url, lastErr)
|
||||
}
|
||||
|
||||
func (c *Client) fetch(ctx context.Context, url string) ([]byte, error) {
|
||||
req, err := http.NewRequestWithContext(ctx, http.MethodGet, url, nil)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
req.Header.Set("User-Agent", c.ua)
|
||||
req.Header.Set("Accept", "text/html,application/xhtml+xml,*/*;q=0.8")
|
||||
req.Header.Set("Accept-Language", "vi,en;q=0.8")
|
||||
|
||||
resp, err := c.http.Do(req)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer func() { _ = resp.Body.Close() }()
|
||||
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
return nil, &statusError{code: resp.StatusCode, status: resp.Status}
|
||||
}
|
||||
return io.ReadAll(io.LimitReader(resp.Body, maxPageSize))
|
||||
}
|
||||
@@ -0,0 +1,198 @@
|
||||
package crawler
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"net/url"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"regexp"
|
||||
"strings"
|
||||
"sync"
|
||||
|
||||
"golang.org/x/sync/errgroup"
|
||||
)
|
||||
|
||||
// Crawler fetches a novel and its chapters.
|
||||
type Crawler struct {
|
||||
Client *Client
|
||||
|
||||
// CacheDir, when set, stores raw pages on disk and serves later runs from
|
||||
// them. Re-exporting with different font or page settings then costs no
|
||||
// requests.
|
||||
CacheDir string
|
||||
|
||||
// Workers bounds concurrent fetches. The client's delay still caps the
|
||||
// overall request rate.
|
||||
Workers int
|
||||
|
||||
// Log receives progress messages. Optional.
|
||||
Log func(format string, args ...any)
|
||||
}
|
||||
|
||||
func (c *Crawler) logf(format string, args ...any) {
|
||||
if c.Log != nil {
|
||||
c.Log(format, args...)
|
||||
}
|
||||
}
|
||||
|
||||
// Novel fetches a novel landing page and resolves its chapter list.
|
||||
//
|
||||
// The chapter list is taken from the landing page and cross-checked against the
|
||||
// dropdown embedded in the first chapter page, so a truncated list cannot
|
||||
// silently shorten the export.
|
||||
func (c *Crawler) Novel(ctx context.Context, novelURL string) (*Novel, error) {
|
||||
page, err := c.page(ctx, novelURL)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
novel, err := ParseNovelPage(page, novelURL)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
c.logf("novel: %s (%d chapters listed)", novel.Title, len(novel.Chapters))
|
||||
|
||||
base, err := url.Parse(novelURL)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
firstPage, err := c.page(ctx, novel.Chapters[0].URL)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
fromSelect, err := ChapterRefsFromSelect(firstPage, base)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
final, extra := ReconcileChapterRefs(novel.Chapters, fromSelect)
|
||||
if len(extra) > 0 {
|
||||
c.logf("warning: %d chapter(s) appear only in the chapter dropdown and were not "+
|
||||
"in the landing page list; verify the export is complete", len(extra))
|
||||
}
|
||||
if len(final) != len(novel.Chapters) {
|
||||
c.logf("chapter list reconciled to %d chapters using the in-chapter dropdown", len(final))
|
||||
}
|
||||
novel.Chapters = final
|
||||
return novel, nil
|
||||
}
|
||||
|
||||
// NovelInfo fetches only the landing page and returns what it carries: title,
|
||||
// slug, tags, and the chapter list as that page shows it.
|
||||
//
|
||||
// Unlike Novel it does not also fetch a chapter page to cross-check the chapter
|
||||
// list, so it costs a single request. Callers that only want metadata should
|
||||
// prefer it; callers about to export every chapter want Novel, whose
|
||||
// reconciliation guards against a silently truncated list.
|
||||
func (c *Crawler) NovelInfo(ctx context.Context, novelURL string) (*Novel, error) {
|
||||
page, err := c.page(ctx, novelURL)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return ParseNovelPage(page, novelURL)
|
||||
}
|
||||
|
||||
// Chapters fetches every chapter concurrently and returns them in reading
|
||||
// order. Any chapter that cannot be fetched or parsed fails the whole run
|
||||
// rather than yielding a book with a hole in it.
|
||||
func (c *Crawler) Chapters(ctx context.Context, novel *Novel) ([]*Chapter, error) {
|
||||
chapters := make([]*Chapter, len(novel.Chapters))
|
||||
|
||||
workers := c.Workers
|
||||
if workers < 1 {
|
||||
workers = 1
|
||||
}
|
||||
|
||||
group, groupCtx := errgroup.WithContext(ctx)
|
||||
group.SetLimit(workers)
|
||||
|
||||
var mu sync.Mutex
|
||||
done := 0
|
||||
|
||||
for i, ref := range novel.Chapters {
|
||||
i, ref := i, ref
|
||||
group.Go(func() error {
|
||||
page, err := c.page(groupCtx, ref.URL)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
chapter, err := ParseChapter(page, ref)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
chapters[i] = chapter
|
||||
|
||||
mu.Lock()
|
||||
done++
|
||||
c.logf("fetched %d/%d: %s (%d words)", done, len(novel.Chapters),
|
||||
chapter.Heading(), chapter.WordCount())
|
||||
mu.Unlock()
|
||||
return nil
|
||||
})
|
||||
}
|
||||
|
||||
if err := group.Wait(); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return chapters, nil
|
||||
}
|
||||
|
||||
// page returns a page from the cache when available, otherwise fetches and
|
||||
// caches it.
|
||||
func (c *Crawler) page(ctx context.Context, pageURL string) ([]byte, error) {
|
||||
path := c.cachePath(pageURL)
|
||||
if path != "" {
|
||||
if body, err := os.ReadFile(path); err == nil && len(body) > 0 { //nolint:gosec // G304: cachePath strips every path separator from the URL
|
||||
return body, nil
|
||||
}
|
||||
}
|
||||
|
||||
body, err := c.Client.Get(ctx, pageURL)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
if path != "" {
|
||||
if err := os.MkdirAll(filepath.Dir(path), 0o750); err == nil {
|
||||
// A failed cache write must not fail the crawl.
|
||||
_ = os.WriteFile(path, body, 0o600)
|
||||
}
|
||||
}
|
||||
return body, nil
|
||||
}
|
||||
|
||||
// unsafeFileChars matches everything not allowed in a cache file name.
|
||||
var unsafeFileChars = regexp.MustCompile(`[^A-Za-z0-9._-]+`)
|
||||
|
||||
// cachePath maps a page URL to a cache file, or "" when caching is disabled.
|
||||
func (c *Crawler) cachePath(pageURL string) string {
|
||||
if c.CacheDir == "" {
|
||||
return ""
|
||||
}
|
||||
u, err := url.Parse(pageURL)
|
||||
if err != nil {
|
||||
return ""
|
||||
}
|
||||
name := unsafeFileChars.ReplaceAllString(strings.Trim(u.Path, "/"), "_")
|
||||
if name == "" {
|
||||
return ""
|
||||
}
|
||||
if !strings.HasSuffix(name, ".html") {
|
||||
name += ".html"
|
||||
}
|
||||
return filepath.Join(c.CacheDir, name)
|
||||
}
|
||||
|
||||
// TotalWords sums the word count across chapters.
|
||||
func TotalWords(chapters []*Chapter) int {
|
||||
n := 0
|
||||
for _, ch := range chapters {
|
||||
n += ch.WordCount()
|
||||
}
|
||||
return n
|
||||
}
|
||||
|
||||
// Describe renders a one-line summary of a crawl result.
|
||||
func Describe(novel *Novel, chapters []*Chapter) string {
|
||||
return fmt.Sprintf("%s — %d chapters, %d words", novel.Title, len(chapters), TotalWords(chapters))
|
||||
}
|
||||
@@ -0,0 +1,50 @@
|
||||
package crawler
|
||||
|
||||
import (
|
||||
"regexp"
|
||||
"strconv"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// The site hides part of every chapter behind CSS rather than putting it in the
|
||||
// markup. Chapter HTML carries empty elements such as
|
||||
//
|
||||
// Nghe <span class="t-3e625e..."></span> trưởng tử
|
||||
//
|
||||
// and the page stylesheet supplies the missing word:
|
||||
//
|
||||
// .t-3e625e...:before { content: "vị"; }
|
||||
//
|
||||
// Reading DOM text alone therefore drops hundreds of words per chapter without
|
||||
// any visible error. wordRule finds those rules so the words can be put back.
|
||||
var wordRule = regexp.MustCompile(`\.([A-Za-z0-9_-]+)\s*::?before\s*\{[^}]*?content\s*:\s*"((?:[^"\\]|\\.)*)"`)
|
||||
|
||||
// cssEscape matches a CSS character escape: a hex code point, optionally
|
||||
// followed by one whitespace terminator, or an escaped literal character.
|
||||
var cssEscape = regexp.MustCompile(`\\([0-9A-Fa-f]{1,6})\s?|\\(.)`)
|
||||
|
||||
// ParseWordClasses maps CSS class name to the word its :before rule injects.
|
||||
func ParseWordClasses(page []byte) map[string]string {
|
||||
words := make(map[string]string)
|
||||
for _, m := range wordRule.FindAllSubmatch(page, -1) {
|
||||
words[string(m[1])] = decodeCSSString(string(m[2]))
|
||||
}
|
||||
return words
|
||||
}
|
||||
|
||||
// decodeCSSString resolves the escape sequences allowed inside a CSS string.
|
||||
func decodeCSSString(s string) string {
|
||||
if !strings.Contains(s, `\`) {
|
||||
return s
|
||||
}
|
||||
return cssEscape.ReplaceAllStringFunc(s, func(esc string) string {
|
||||
m := cssEscape.FindStringSubmatch(esc)
|
||||
if m[1] != "" {
|
||||
if cp, err := strconv.ParseInt(m[1], 16, 32); err == nil && cp > 0 {
|
||||
return string(rune(cp))
|
||||
}
|
||||
return ""
|
||||
}
|
||||
return m[2]
|
||||
})
|
||||
}
|
||||
@@ -0,0 +1,95 @@
|
||||
package crawler
|
||||
|
||||
import (
|
||||
"strings"
|
||||
|
||||
"golang.org/x/net/html"
|
||||
)
|
||||
|
||||
// attr returns the value of the named attribute, or "" when absent.
|
||||
func attr(n *html.Node, name string) string {
|
||||
for _, a := range n.Attr {
|
||||
if a.Key == name {
|
||||
return a.Val
|
||||
}
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
// hasClass reports whether the node carries the given class token.
|
||||
func hasClass(n *html.Node, class string) bool {
|
||||
for _, tok := range strings.Fields(attr(n, "class")) {
|
||||
if tok == class {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// findNode returns the first node in document order satisfying match.
|
||||
func findNode(root *html.Node, match func(*html.Node) bool) *html.Node {
|
||||
if match(root) {
|
||||
return root
|
||||
}
|
||||
for c := root.FirstChild; c != nil; c = c.NextSibling {
|
||||
if found := findNode(c, match); found != nil {
|
||||
return found
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// findAllNodes returns every node satisfying match, in document order.
|
||||
func findAllNodes(root *html.Node, match func(*html.Node) bool) []*html.Node {
|
||||
var out []*html.Node
|
||||
var walk func(*html.Node)
|
||||
walk = func(n *html.Node) {
|
||||
if match(n) {
|
||||
out = append(out, n)
|
||||
}
|
||||
for c := n.FirstChild; c != nil; c = c.NextSibling {
|
||||
walk(c)
|
||||
}
|
||||
}
|
||||
walk(root)
|
||||
return out
|
||||
}
|
||||
|
||||
// elementByID finds an element by its id attribute.
|
||||
func elementByID(root *html.Node, id string) *html.Node {
|
||||
return findNode(root, func(n *html.Node) bool {
|
||||
return n.Type == html.ElementNode && attr(n, "id") == id
|
||||
})
|
||||
}
|
||||
|
||||
// elementByTag finds the first element with the given tag name.
|
||||
func elementByTag(root *html.Node, tag string) *html.Node {
|
||||
return findNode(root, func(n *html.Node) bool {
|
||||
return n.Type == html.ElementNode && n.Data == tag
|
||||
})
|
||||
}
|
||||
|
||||
// nodeText collects the descendant text of a node with whitespace collapsed.
|
||||
func nodeText(n *html.Node) string {
|
||||
if n == nil {
|
||||
return ""
|
||||
}
|
||||
var b strings.Builder
|
||||
var walk func(*html.Node)
|
||||
walk = func(n *html.Node) {
|
||||
if n.Type == html.TextNode {
|
||||
b.WriteString(n.Data)
|
||||
}
|
||||
for c := n.FirstChild; c != nil; c = c.NextSibling {
|
||||
walk(c)
|
||||
}
|
||||
}
|
||||
walk(n)
|
||||
return collapseSpaces(b.String())
|
||||
}
|
||||
|
||||
// collapseSpaces trims the string and reduces every whitespace run, including
|
||||
// the non-breaking spaces the site emits as , to a single space.
|
||||
func collapseSpaces(s string) string {
|
||||
return strings.Join(strings.Fields(s), " ")
|
||||
}
|
||||
@@ -0,0 +1,223 @@
|
||||
package crawler
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"fmt"
|
||||
"net/url"
|
||||
"strings"
|
||||
|
||||
"golang.org/x/net/html"
|
||||
)
|
||||
|
||||
// ChapterRef points at a single chapter listed on a novel page.
|
||||
type ChapterRef struct {
|
||||
Label string // as shown on the site, e.g. "14" or "Chương 12"
|
||||
URL string
|
||||
}
|
||||
|
||||
// Novel is a novel's landing page: its title and its chapters in reading order.
|
||||
type Novel struct {
|
||||
Title string
|
||||
Slug string
|
||||
URL string
|
||||
|
||||
// Tags are the novel's own genres, as labelled on the site, in the order
|
||||
// they appear. Empty when the page lists none.
|
||||
Tags []string
|
||||
Chapters []ChapterRef
|
||||
}
|
||||
|
||||
// ParseNovelPage reads the title and chapter list from a novel landing page.
|
||||
//
|
||||
// Chapter URLs are always taken from the anchors on the page. Slugs are not
|
||||
// uniform across novels ("/14.html" on one, "/chuong-12.html" on another) and
|
||||
// numbering has gaps, so generating URLs from a chapter count would fetch 404s
|
||||
// and miss real chapters.
|
||||
func ParseNovelPage(page []byte, pageURL string) (*Novel, error) {
|
||||
doc, err := html.Parse(bytes.NewReader(page))
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("parse novel page: %w", err)
|
||||
}
|
||||
base, err := url.Parse(pageURL)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("parse novel url: %w", err)
|
||||
}
|
||||
|
||||
novel := &Novel{
|
||||
Title: novelTitle(doc),
|
||||
Slug: slugFromNovelURL(base),
|
||||
URL: pageURL,
|
||||
Tags: tagsFromInfo(doc),
|
||||
Chapters: chapterRefsFromList(doc, base),
|
||||
}
|
||||
if novel.Title == "" {
|
||||
novel.Title = novel.Slug
|
||||
}
|
||||
if len(novel.Chapters) == 0 {
|
||||
return nil, fmt.Errorf("no chapters found on %s (page layout may have changed)", pageURL)
|
||||
}
|
||||
return novel, nil
|
||||
}
|
||||
|
||||
// novelTitle prefers the <h1> heading and falls back to the document title.
|
||||
func novelTitle(doc *html.Node) string {
|
||||
if h1 := elementByTag(doc, "h1"); h1 != nil {
|
||||
if t := nodeText(h1); t != "" {
|
||||
return t
|
||||
}
|
||||
}
|
||||
return nodeText(elementByTag(doc, "title"))
|
||||
}
|
||||
|
||||
// tagsFromInfo reads the novel's own genres out of the info block.
|
||||
//
|
||||
// The anchors are matched on the schema.org itemprop="genre" microdata rather
|
||||
// than on their href. The page also carries a site-wide genre menu linking every
|
||||
// category — 69 distinct ones against this novel's 6 on a sampled page — so
|
||||
// matching the /the-loai/ URL shape would pull in the whole navigation.
|
||||
func tagsFromInfo(doc *html.Node) []string {
|
||||
links := findAllNodes(doc, func(n *html.Node) bool {
|
||||
return n.Type == html.ElementNode && n.Data == "a" && attr(n, "itemprop") == "genre"
|
||||
})
|
||||
|
||||
var tags []string
|
||||
seen := make(map[string]bool, len(links))
|
||||
for _, link := range links {
|
||||
// The label appears both as the link text and as its title attribute;
|
||||
// prefer the text and fall back to the attribute.
|
||||
name := nodeText(link)
|
||||
if name == "" {
|
||||
name = collapseSpaces(attr(link, "title"))
|
||||
}
|
||||
if name == "" || seen[name] {
|
||||
continue
|
||||
}
|
||||
seen[name] = true
|
||||
tags = append(tags, name)
|
||||
}
|
||||
return tags
|
||||
}
|
||||
|
||||
// slugFromNovelURL turns https://host/tro-lai-nam-thang-cu.html into
|
||||
// "tro-lai-nam-thang-cu".
|
||||
func slugFromNovelURL(u *url.URL) string {
|
||||
seg := strings.Trim(u.Path, "/")
|
||||
if i := strings.LastIndex(seg, "/"); i >= 0 {
|
||||
seg = seg[i+1:]
|
||||
}
|
||||
return strings.TrimSuffix(seg, ".html")
|
||||
}
|
||||
|
||||
// chapterRefsFromList reads the "list-chapters" block on the landing page.
|
||||
// The site lists newest first, so the result is reversed into reading order.
|
||||
func chapterRefsFromList(doc *html.Node, base *url.URL) []ChapterRef {
|
||||
list := findNode(doc, func(n *html.Node) bool {
|
||||
return n.Type == html.ElementNode && hasClass(n, "list-chapters")
|
||||
})
|
||||
if list == nil {
|
||||
return nil
|
||||
}
|
||||
|
||||
titles := findAllNodes(list, func(n *html.Node) bool {
|
||||
return n.Type == html.ElementNode && hasClass(n, "episode-title")
|
||||
})
|
||||
|
||||
var refs []ChapterRef
|
||||
for _, title := range titles {
|
||||
link := elementByTag(title, "a")
|
||||
if link == nil {
|
||||
continue
|
||||
}
|
||||
href := strings.TrimSpace(attr(link, "href"))
|
||||
if href == "" {
|
||||
continue
|
||||
}
|
||||
abs, err := base.Parse(href)
|
||||
if err != nil {
|
||||
continue
|
||||
}
|
||||
refs = append(refs, ChapterRef{Label: nodeText(link), URL: abs.String()})
|
||||
}
|
||||
return reverseRefs(refs)
|
||||
}
|
||||
|
||||
// ChapterRefsFromSelect reads the chapter dropdown embedded in every chapter
|
||||
// page, whose options hold "novel-slug,chapter-slug" pairs. This is a second,
|
||||
// independent view of the chapter list used to cross-check the landing page.
|
||||
func ChapterRefsFromSelect(page []byte, base *url.URL) ([]ChapterRef, error) {
|
||||
doc, err := html.Parse(bytes.NewReader(page))
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("parse chapter page: %w", err)
|
||||
}
|
||||
sel := elementByID(doc, "selected_chapter")
|
||||
if sel == nil {
|
||||
return nil, nil
|
||||
}
|
||||
|
||||
var refs []ChapterRef
|
||||
for _, opt := range findAllNodes(sel, func(n *html.Node) bool {
|
||||
return n.Type == html.ElementNode && n.Data == "option"
|
||||
}) {
|
||||
novelSlug, chapterSlug, ok := strings.Cut(attr(opt, "value"), ",")
|
||||
if !ok || novelSlug == "" || chapterSlug == "" {
|
||||
continue
|
||||
}
|
||||
abs, err := base.Parse("/" + novelSlug + "/" + chapterSlug + ".html")
|
||||
if err != nil {
|
||||
continue
|
||||
}
|
||||
refs = append(refs, ChapterRef{Label: nodeText(opt), URL: abs.String()})
|
||||
}
|
||||
return reverseRefs(refs), nil
|
||||
}
|
||||
|
||||
// ReconcileChapterRefs picks the chapter list to crawl from the landing page
|
||||
// list and the in-chapter dropdown.
|
||||
//
|
||||
// Both views come from the same site ordering, so when the dropdown is a
|
||||
// superset it is preferred: that keeps the run correct even if the landing page
|
||||
// ever truncates or paginates its list. Anything the dropdown alone knows about
|
||||
// while disagreeing on order is returned as extra so the caller can warn rather
|
||||
// than silently export a short book.
|
||||
func ReconcileChapterRefs(fromList, fromSelect []ChapterRef) (final, extra []ChapterRef) {
|
||||
if len(fromSelect) == 0 {
|
||||
return fromList, nil
|
||||
}
|
||||
|
||||
inSelect := refURLSet(fromSelect)
|
||||
if listIsSubsetOf(fromList, inSelect) {
|
||||
return fromSelect, nil
|
||||
}
|
||||
|
||||
inList := refURLSet(fromList)
|
||||
for _, ref := range fromSelect {
|
||||
if !inList[ref.URL] {
|
||||
extra = append(extra, ref)
|
||||
}
|
||||
}
|
||||
return fromList, extra
|
||||
}
|
||||
|
||||
func refURLSet(refs []ChapterRef) map[string]bool {
|
||||
set := make(map[string]bool, len(refs))
|
||||
for _, r := range refs {
|
||||
set[r.URL] = true
|
||||
}
|
||||
return set
|
||||
}
|
||||
|
||||
func listIsSubsetOf(refs []ChapterRef, set map[string]bool) bool {
|
||||
for _, r := range refs {
|
||||
if !set[r.URL] {
|
||||
return false
|
||||
}
|
||||
}
|
||||
return true
|
||||
}
|
||||
|
||||
func reverseRefs(refs []ChapterRef) []ChapterRef {
|
||||
for i, j := 0, len(refs)-1; i < j; i, j = i+1, j-1 {
|
||||
refs[i], refs[j] = refs[j], refs[i]
|
||||
}
|
||||
return refs
|
||||
}
|
||||
@@ -0,0 +1,170 @@
|
||||
package crawler
|
||||
|
||||
import (
|
||||
"net/url"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// novelFixture reproduces the two traits that break naive crawlers: the list is
|
||||
// newest-first, and chapter numbering has a gap (no chapter 4).
|
||||
const novelFixture = `<html><head><title>TEN TRUYEN</title></head><body>
|
||||
<h1>TEN TRUYEN</h1>
|
||||
<div class="list-chapters">
|
||||
<div class="item"><div class="episode-title"><a href="https://monkeydd.com/n/5.html">5</a></div></div>
|
||||
<div class="item"><div class="episode-title"><a href="https://monkeydd.com/n/3.html">3</a></div></div>
|
||||
<div class="item"><div class="episode-title"><a href="/n/2.html">2</a></div></div>
|
||||
<div class="item"><div class="episode-title"><a href="https://monkeydd.com/n/1.html">1</a></div></div>
|
||||
</div></body></html>`
|
||||
|
||||
func TestParseNovelPageOrdersChaptersForReading(t *testing.T) {
|
||||
novel, err := ParseNovelPage([]byte(novelFixture), "https://monkeydd.com/n.html")
|
||||
if err != nil {
|
||||
t.Fatalf("ParseNovelPage: %v", err)
|
||||
}
|
||||
|
||||
if novel.Title != "TEN TRUYEN" {
|
||||
t.Errorf("Title = %q", novel.Title)
|
||||
}
|
||||
if novel.Slug != "n" {
|
||||
t.Errorf("Slug = %q, want %q", novel.Slug, "n")
|
||||
}
|
||||
|
||||
wantLabels := []string{"1", "2", "3", "5"}
|
||||
if len(novel.Chapters) != len(wantLabels) {
|
||||
t.Fatalf("got %d chapters, want %d", len(novel.Chapters), len(wantLabels))
|
||||
}
|
||||
for i, want := range wantLabels {
|
||||
if novel.Chapters[i].Label != want {
|
||||
t.Errorf("chapter %d label = %q, want %q", i, novel.Chapters[i].Label, want)
|
||||
}
|
||||
}
|
||||
// Relative hrefs must resolve against the novel URL.
|
||||
if got, want := novel.Chapters[1].URL, "https://monkeydd.com/n/2.html"; got != want {
|
||||
t.Errorf("chapter 2 URL = %q, want %q", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
// tagsFixture mirrors the real page: the novel's own genres carry
|
||||
// itemprop="genre", while a site-wide menu links every category without it. A
|
||||
// parser matching on the /the-loai/ href shape would swallow the whole menu.
|
||||
const tagsFixture = `<html><head><title>TEN TRUYEN</title></head><body>
|
||||
<h1>TEN TRUYEN</h1>
|
||||
<dl class="row">
|
||||
<dt class="col-sm-3">Thể loại</dt>
|
||||
<dd class="col-sm-9">
|
||||
<a class='cate-item' itemprop='genre' title='Trọng Sinh' href='https://monkeydd.com/the-loai/trong-sinh.html'>Trọng Sinh</a>
|
||||
<a class='cate-item' itemprop='genre' title='Cổ Đại' href='https://monkeydd.com/the-loai/co-dai.html'>Cổ Đại</a>
|
||||
<a class='cate-item' itemprop='genre' title='Gia Đình' href='https://monkeydd.com/the-loai/gia-dinh.html'></a>
|
||||
<a class='cate-item' itemprop='genre' title='Cổ Đại' href='https://monkeydd.com/the-loai/co-dai.html'>Cổ Đại</a>
|
||||
</dd>
|
||||
</dl>
|
||||
<nav class="site-menu">
|
||||
<a href='https://monkeydd.com/the-loai/dam-my.html'>Đam Mỹ</a>
|
||||
<a href='https://monkeydd.com/the-loai/bach-hop.html'>Bách Hợp</a>
|
||||
</nav>
|
||||
<div class="list-chapters">
|
||||
<div class="item"><div class="episode-title"><a href="/n/1.html">1</a></div></div>
|
||||
</div></body></html>`
|
||||
|
||||
func TestParseNovelPageReadsTags(t *testing.T) {
|
||||
novel, err := ParseNovelPage([]byte(tagsFixture), "https://monkeydd.com/n.html")
|
||||
if err != nil {
|
||||
t.Fatalf("ParseNovelPage: %v", err)
|
||||
}
|
||||
|
||||
// "Cổ Đại" has its inner whitespace collapsed, the empty anchor falls back
|
||||
// to its title attribute, and the repeated genre appears once.
|
||||
want := []string{"Trọng Sinh", "Cổ Đại", "Gia Đình"}
|
||||
if len(novel.Tags) != len(want) {
|
||||
t.Fatalf("Tags = %q, want %q", novel.Tags, want)
|
||||
}
|
||||
for i := range want {
|
||||
if novel.Tags[i] != want[i] {
|
||||
t.Errorf("Tags[%d] = %q, want %q", i, novel.Tags[i], want[i])
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestParseNovelPageWithoutTags(t *testing.T) {
|
||||
novel, err := ParseNovelPage([]byte(novelFixture), "https://monkeydd.com/n.html")
|
||||
if err != nil {
|
||||
t.Fatalf("ParseNovelPage: %v", err)
|
||||
}
|
||||
if len(novel.Tags) != 0 {
|
||||
t.Errorf("Tags = %q, want none", novel.Tags)
|
||||
}
|
||||
}
|
||||
|
||||
func TestParseNovelPageNoChapters(t *testing.T) {
|
||||
if _, err := ParseNovelPage([]byte(`<html><title>x</title><body></body></html>`),
|
||||
"https://monkeydd.com/n.html"); err == nil {
|
||||
t.Fatal("want an error when no chapters are listed")
|
||||
}
|
||||
}
|
||||
|
||||
const selectFixture = `<html><body>
|
||||
<select name="selected_chapter" id="selected_chapter">
|
||||
<option value="n,5">5</option>
|
||||
<option value="n,chuong-3">Chương 3</option>
|
||||
<option value="n,1">1</option>
|
||||
<option value="14">malformed</option>
|
||||
</select></body></html>`
|
||||
|
||||
func TestChapterRefsFromSelect(t *testing.T) {
|
||||
base, err := url.Parse("https://monkeydd.com/n.html")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
refs, err := ChapterRefsFromSelect([]byte(selectFixture), base)
|
||||
if err != nil {
|
||||
t.Fatalf("ChapterRefsFromSelect: %v", err)
|
||||
}
|
||||
|
||||
if len(refs) != 3 {
|
||||
t.Fatalf("got %d refs %+v, want 3 (malformed option skipped)", len(refs), refs)
|
||||
}
|
||||
if got, want := refs[0].URL, "https://monkeydd.com/n/1.html"; got != want {
|
||||
t.Errorf("first ref URL = %q, want %q", got, want)
|
||||
}
|
||||
if got, want := refs[1].URL, "https://monkeydd.com/n/chuong-3.html"; got != want {
|
||||
t.Errorf("second ref URL = %q, want %q", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestReconcileChapterRefs(t *testing.T) {
|
||||
ref := func(u string) ChapterRef { return ChapterRef{Label: u, URL: u} }
|
||||
|
||||
t.Run("dropdown superset wins", func(t *testing.T) {
|
||||
list := []ChapterRef{ref("a"), ref("b")}
|
||||
sel := []ChapterRef{ref("a"), ref("b"), ref("c")}
|
||||
|
||||
final, extra := ReconcileChapterRefs(list, sel)
|
||||
if len(final) != 3 {
|
||||
t.Errorf("got %d chapters, want the 3 from the dropdown", len(final))
|
||||
}
|
||||
if len(extra) != 0 {
|
||||
t.Errorf("got %d extra, want 0", len(extra))
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("disagreement is reported not silently merged", func(t *testing.T) {
|
||||
list := []ChapterRef{ref("a"), ref("z")}
|
||||
sel := []ChapterRef{ref("a"), ref("b")}
|
||||
|
||||
final, extra := ReconcileChapterRefs(list, sel)
|
||||
if len(final) != 2 || final[1].URL != "z" {
|
||||
t.Errorf("final = %+v, want the landing page list", final)
|
||||
}
|
||||
if len(extra) != 1 || extra[0].URL != "b" {
|
||||
t.Errorf("extra = %+v, want [b] so the caller can warn", extra)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("empty dropdown falls back to the list", func(t *testing.T) {
|
||||
list := []ChapterRef{ref("a")}
|
||||
final, extra := ReconcileChapterRefs(list, nil)
|
||||
if len(final) != 1 || len(extra) != 0 {
|
||||
t.Errorf("final = %+v, extra = %+v", final, extra)
|
||||
}
|
||||
})
|
||||
}
|
||||
@@ -0,0 +1,154 @@
|
||||
// Package export turns a novel URL into a PDF file: resolve the chapter list,
|
||||
// fetch the chapters, render the PDF with the bundled font.
|
||||
package export
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"net/url"
|
||||
"path/filepath"
|
||||
"regexp"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"github.com/tiennm99/miti99bot/internal/modules/monkeyd/crawler"
|
||||
"github.com/tiennm99/miti99bot/internal/modules/monkeyd/pdf"
|
||||
)
|
||||
|
||||
// Defaults the bot advertises to users or reuses in other crawls.
|
||||
const (
|
||||
DefaultFontSize = 10.0
|
||||
DefaultDelay = 400 * time.Millisecond
|
||||
)
|
||||
|
||||
// Fixed layout and crawl settings. The bot exposes only the font size.
|
||||
const (
|
||||
lineSpacing = 1.55 // multiple of font size
|
||||
margin = 6.0 // millimetres
|
||||
workers = 4
|
||||
retries = 3
|
||||
)
|
||||
|
||||
// Request describes one export.
|
||||
type Request struct {
|
||||
NovelURL string
|
||||
|
||||
// OutDir receives the PDF, named after the novel title.
|
||||
OutDir string
|
||||
|
||||
// FontSize is in points. 0 means DefaultFontSize.
|
||||
FontSize float64
|
||||
|
||||
// CacheDir stores raw pages so a re-export costs no requests. Empty
|
||||
// disables the cache.
|
||||
CacheDir string
|
||||
|
||||
// Log receives progress messages. Optional.
|
||||
Log func(format string, args ...any)
|
||||
}
|
||||
|
||||
// Result reports what was produced.
|
||||
type Result struct {
|
||||
Path string
|
||||
Title string
|
||||
SourceURL string
|
||||
Chapters int
|
||||
Words int
|
||||
Page pdf.PageSize
|
||||
}
|
||||
|
||||
// Summary renders a one-line description of the exported book.
|
||||
func (r *Result) Summary() string {
|
||||
return fmt.Sprintf("%s — %d chapters, %d words", r.Title, r.Chapters, r.Words)
|
||||
}
|
||||
|
||||
// Export fetches the novel at req.NovelURL and writes it as a PDF, returning
|
||||
// where it landed. The context bounds the whole crawl; cancelling it abandons
|
||||
// the run without leaving a partial PDF behind.
|
||||
func Export(ctx context.Context, req Request) (*Result, error) {
|
||||
if req.FontSize == 0 {
|
||||
req.FontSize = DefaultFontSize
|
||||
}
|
||||
if err := req.validate(); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
c := &crawler.Crawler{
|
||||
Client: crawler.NewClient(DefaultDelay, retries),
|
||||
CacheDir: req.CacheDir,
|
||||
Workers: workers,
|
||||
Log: req.Log,
|
||||
}
|
||||
|
||||
novel, err := c.Novel(ctx, req.NovelURL)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
chapters, err := c.Chapters(ctx, novel)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
outPath := filepath.Join(req.OutDir, SafeFileName(novel.Title, novel.Slug)+".pdf")
|
||||
opts := pdf.Options{
|
||||
Page: pdf.PhonePage,
|
||||
Margin: margin,
|
||||
Font: pdf.BundledFont(),
|
||||
FontSize: req.FontSize,
|
||||
LineSpacing: lineSpacing,
|
||||
Title: novel.Title,
|
||||
SourceURL: novel.URL,
|
||||
}
|
||||
if err := pdf.Write(outPath, opts, toPDFChapters(chapters)); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
return &Result{
|
||||
Path: outPath,
|
||||
Title: novel.Title,
|
||||
SourceURL: novel.URL,
|
||||
Chapters: len(chapters),
|
||||
Words: crawler.TotalWords(chapters),
|
||||
Page: pdf.PhonePage,
|
||||
}, nil
|
||||
}
|
||||
|
||||
// validate rejects a Request before any request is made, so a typo costs no
|
||||
// fetches.
|
||||
func (r *Request) validate() error {
|
||||
if r.NovelURL == "" {
|
||||
return fmt.Errorf("novel url is required")
|
||||
}
|
||||
parsed, err := url.Parse(r.NovelURL)
|
||||
if err != nil {
|
||||
return fmt.Errorf("invalid novel url: %w", err)
|
||||
}
|
||||
if parsed.Scheme != "http" && parsed.Scheme != "https" {
|
||||
return fmt.Errorf("invalid novel url: want an http(s) URL, got %q", r.NovelURL)
|
||||
}
|
||||
if r.FontSize <= 0 {
|
||||
return fmt.Errorf("font size must be positive")
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func toPDFChapters(chapters []*crawler.Chapter) []pdf.Chapter {
|
||||
out := make([]pdf.Chapter, 0, len(chapters))
|
||||
for _, ch := range chapters {
|
||||
out = append(out, pdf.Chapter{Heading: ch.Heading(), Paragraphs: ch.Paragraphs})
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
var unsafeNameChars = regexp.MustCompile(`[^\p{L}\p{N}]+`)
|
||||
|
||||
// SafeFileName builds a file name from the novel title, falling back to the
|
||||
// slug when the title has no usable characters. The result has no extension.
|
||||
func SafeFileName(title, fallback string) string {
|
||||
name := strings.Trim(unsafeNameChars.ReplaceAllString(title, "-"), "-")
|
||||
if name == "" {
|
||||
return fallback
|
||||
}
|
||||
return name
|
||||
}
|
||||
@@ -0,0 +1,81 @@
|
||||
package export
|
||||
|
||||
import "testing"
|
||||
|
||||
func TestValidate(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
req Request
|
||||
wantErr bool
|
||||
}{
|
||||
{
|
||||
name: "url and font size are valid",
|
||||
req: Request{NovelURL: "https://monkeydd.com/example.html", FontSize: DefaultFontSize},
|
||||
},
|
||||
{
|
||||
name: "missing url",
|
||||
req: Request{FontSize: DefaultFontSize},
|
||||
wantErr: true,
|
||||
},
|
||||
{
|
||||
name: "non-http scheme",
|
||||
req: Request{NovelURL: "ftp://monkeydd.com/example.html", FontSize: DefaultFontSize},
|
||||
wantErr: true,
|
||||
},
|
||||
{
|
||||
name: "scheme-less url",
|
||||
req: Request{NovelURL: "monkeydd.com/example.html", FontSize: DefaultFontSize},
|
||||
wantErr: true,
|
||||
},
|
||||
{
|
||||
name: "negative font size",
|
||||
req: Request{NovelURL: "https://monkeydd.com/example.html", FontSize: -1},
|
||||
wantErr: true,
|
||||
},
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
err := tt.req.validate()
|
||||
if tt.wantErr && err == nil {
|
||||
t.Error("validate() = nil, want error")
|
||||
}
|
||||
if !tt.wantErr && err != nil {
|
||||
t.Errorf("validate() = %v, want nil", err)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// An invalid request must fail before the crawler makes any request.
|
||||
func TestExportRejectsInvalidRequestWithoutFetching(t *testing.T) {
|
||||
if _, err := Export(t.Context(), Request{NovelURL: "ftp://monkeydd.com/x.html", OutDir: t.TempDir()}); err == nil {
|
||||
t.Fatal("Export = nil error, want an invalid url error")
|
||||
}
|
||||
}
|
||||
|
||||
func TestSafeFileName(t *testing.T) {
|
||||
tests := []struct {
|
||||
title string
|
||||
fallback string
|
||||
want string
|
||||
}{
|
||||
{"Trở Lại Năm Tháng Cũ", "slug", "Trở-Lại-Năm-Tháng-Cũ"},
|
||||
{"Chapter: One / Two", "slug", "Chapter-One-Two"},
|
||||
{" --- ", "slug", "slug"},
|
||||
{"", "slug", "slug"},
|
||||
}
|
||||
for _, tt := range tests {
|
||||
if got := SafeFileName(tt.title, tt.fallback); got != tt.want {
|
||||
t.Errorf("SafeFileName(%q, %q) = %q, want %q", tt.title, tt.fallback, got, tt.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestResultSummary(t *testing.T) {
|
||||
r := &Result{Title: "Example", Chapters: 12, Words: 3400}
|
||||
want := "Example — 12 chapters, 3400 words"
|
||||
if got := r.Summary(); got != want {
|
||||
t.Errorf("Summary() = %q, want %q", got, want)
|
||||
}
|
||||
}
|
||||
@@ -11,7 +11,7 @@ import (
|
||||
"github.com/go-telegram/bot"
|
||||
"github.com/go-telegram/bot/models"
|
||||
|
||||
"github.com/tiennm99/monkeyd-crawler/export"
|
||||
"github.com/tiennm99/miti99bot/internal/modules/monkeyd/export"
|
||||
|
||||
"github.com/tiennm99/miti99bot/internal/log"
|
||||
"github.com/tiennm99/miti99bot/internal/modules/util/chathelper"
|
||||
|
||||
@@ -9,8 +9,8 @@ import (
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"github.com/tiennm99/monkeyd-crawler/export"
|
||||
"github.com/tiennm99/monkeyd-crawler/pdfout"
|
||||
"github.com/tiennm99/miti99bot/internal/modules/monkeyd/export"
|
||||
"github.com/tiennm99/miti99bot/internal/modules/monkeyd/pdf"
|
||||
|
||||
"github.com/tiennm99/miti99bot/internal/modules"
|
||||
"github.com/tiennm99/miti99bot/internal/storage"
|
||||
@@ -77,7 +77,7 @@ func stubPDF(t *testing.T, dir, name string, size int64) *export.Result {
|
||||
SourceURL: testNovelURL,
|
||||
Chapters: 3,
|
||||
Words: 1200,
|
||||
Page: pdfout.Presets["phone"],
|
||||
Page: pdf.PhonePage,
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -15,7 +15,7 @@ import (
|
||||
"github.com/go-telegram/bot"
|
||||
"github.com/go-telegram/bot/models"
|
||||
|
||||
"github.com/tiennm99/monkeyd-crawler/export"
|
||||
"github.com/tiennm99/miti99bot/internal/modules/monkeyd/export"
|
||||
|
||||
"github.com/tiennm99/miti99bot/internal/modules"
|
||||
"github.com/tiennm99/miti99bot/internal/modules/util/chathelper"
|
||||
|
||||
@@ -0,0 +1,37 @@
|
||||
package pdf
|
||||
|
||||
import (
|
||||
_ "embed"
|
||||
)
|
||||
|
||||
// bundledTTF is the only font the bot renders with. A minimal container has no
|
||||
// system fonts, and a single embedded font keeps the PDF identical on every
|
||||
// host. See fonts/NOTICE.md for provenance and licensing.
|
||||
//
|
||||
// Vietnamese text needs the Latin Extended Additional block (ư, ạ, ế, ộ …);
|
||||
// DejaVu Sans covers it, where a basic-Latin font would silently drop the
|
||||
// diacritics.
|
||||
//
|
||||
//go:embed fonts/DejaVuSans.ttf
|
||||
var bundledTTF []byte
|
||||
|
||||
// BundledFontName labels the embedded font in diagnostics. It is not a path;
|
||||
// the font is compiled into the binary.
|
||||
const BundledFontName = "DejaVu Sans (bundled)"
|
||||
|
||||
// Font is font data ready to embed in a PDF.
|
||||
//
|
||||
// The data is carried as bytes rather than as a path because fpdf joins a font
|
||||
// path onto its own font directory, which it defaults to "." — turning an
|
||||
// absolute path into a working-directory-relative one that resolves only when
|
||||
// the process happens to run from the filesystem root.
|
||||
type Font struct {
|
||||
// Name identifies the font for diagnostics.
|
||||
Name string
|
||||
Data []byte
|
||||
}
|
||||
|
||||
// BundledFont returns the font compiled into the binary.
|
||||
func BundledFont() Font {
|
||||
return Font{Name: BundledFontName, Data: bundledTTF}
|
||||
}
|
||||
@@ -0,0 +1,38 @@
|
||||
package pdf
|
||||
|
||||
import (
|
||||
"testing"
|
||||
|
||||
"golang.org/x/image/font/sfnt"
|
||||
)
|
||||
|
||||
func TestBundledFontIsParseable(t *testing.T) {
|
||||
font := BundledFont()
|
||||
if font.Name != BundledFontName {
|
||||
t.Errorf("Name = %q, want %q", font.Name, BundledFontName)
|
||||
}
|
||||
if _, err := sfnt.Parse(font.Data); err != nil {
|
||||
t.Fatalf("bundled font does not parse as a TrueType font: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
// The bundled font exists to render Vietnamese. A replacement that lacks these
|
||||
// glyphs would silently emit blanks, so check before trusting it.
|
||||
func TestBundledFontCoversVietnamese(t *testing.T) {
|
||||
parsed, err := sfnt.Parse(BundledFont().Data)
|
||||
if err != nil {
|
||||
t.Fatalf("parse bundled font: %v", err)
|
||||
}
|
||||
var buf sfnt.Buffer
|
||||
for _, r := range []rune{'ư', 'ơ', 'đ', 'ạ', 'ế', 'ộ', 'ữ', 'ằ', 'ỷ', 'ỹ', 'Ọ', 'Ế', 'Ư', 'Đ'} {
|
||||
index, err := parsed.GlyphIndex(&buf, r)
|
||||
if err != nil {
|
||||
t.Errorf("GlyphIndex(%q): %v", r, err)
|
||||
continue
|
||||
}
|
||||
// Glyph 0 is .notdef — the character is absent from the font.
|
||||
if index == 0 {
|
||||
t.Errorf("bundled font has no glyph for %q (U+%04X)", r, r)
|
||||
}
|
||||
}
|
||||
}
|
||||
Binary file not shown.
@@ -0,0 +1,24 @@
|
||||
# Bundled font
|
||||
|
||||
`DejaVuSans.ttf` is embedded into the binary and used when no font is supplied
|
||||
and no suitable system font is found. It covers the Latin Extended Additional
|
||||
block, which is what Vietnamese diacritics need — a font with only basic Latin
|
||||
coverage silently drops them.
|
||||
|
||||
The following is recorded in the font file's own name table:
|
||||
|
||||
- Version: `Version 2.37`
|
||||
- Copyright: `Copyright (c) 2003 by Bitstream, Inc. All Rights Reserved.`
|
||||
`Copyright (c) 2006 by Tavmjong Bah. All Rights Reserved.`
|
||||
`DejaVu changes are in public domain`
|
||||
- License information: <http://dejavu.sourceforge.net/wiki/index.php/License>
|
||||
|
||||
The DejaVu fonts are free and redistributable, which is why they ship with most
|
||||
Linux distributions. This copy came from Alpine's `font-dejavu` package, which
|
||||
does not include the license text as a separate file. For a vendored copy of the
|
||||
full license text, take `LICENSE` from the upstream DejaVu release and add it to
|
||||
this directory.
|
||||
|
||||
To swap the bundled font, replace `DejaVuSans.ttf` and update this file. Verify
|
||||
the replacement covers Vietnamese first — `TestBundledFontCoversVietnamese` in
|
||||
`../font_test.go` checks a representative set of characters.
|
||||
@@ -0,0 +1,132 @@
|
||||
package pdf
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"strings"
|
||||
|
||||
"github.com/go-pdf/fpdf"
|
||||
)
|
||||
|
||||
// pointsToMM converts typographic points to millimetres.
|
||||
const pointsToMM = 25.4 / 72.0
|
||||
|
||||
// bodyFont is the internal family name registered with the PDF.
|
||||
const bodyFont = "body"
|
||||
|
||||
// PageSize is a page in millimetres.
|
||||
type PageSize struct {
|
||||
Name string
|
||||
W, H float64
|
||||
}
|
||||
|
||||
// PhonePage is the page every export uses.
|
||||
//
|
||||
// Phone reading depends far more on page shape than on font size. A phone
|
||||
// viewer scales a whole page to fit the screen, so a large font on an A4 page
|
||||
// still ends up tiny: the page is ~3x wider than the screen and gets shrunk to
|
||||
// match. A page cut to the phone's own aspect ratio fills the screen at 100%,
|
||||
// which is why this is a small 9:16 page rather than A4 with big type.
|
||||
var PhonePage = PageSize{Name: "phone", W: 90, H: 160}
|
||||
|
||||
// Chapter is a chapter ready to render.
|
||||
type Chapter struct {
|
||||
Heading string
|
||||
Paragraphs []string
|
||||
}
|
||||
|
||||
// Options controls the exported PDF.
|
||||
type Options struct {
|
||||
Page PageSize
|
||||
Margin float64 // mm
|
||||
Font Font // BundledFont()
|
||||
FontSize float64 // pt
|
||||
LineSpacing float64 // multiple of font size
|
||||
Title string
|
||||
SourceURL string
|
||||
}
|
||||
|
||||
// footerReserve is the vertical space kept clear for the page number.
|
||||
const footerReserve = 6.0
|
||||
|
||||
// Write renders the chapters to a PDF at path.
|
||||
func Write(path string, opts Options, chapters []Chapter) error {
|
||||
pdf := fpdf.NewCustom(&fpdf.InitType{
|
||||
UnitStr: "mm",
|
||||
Size: fpdf.SizeType{Wd: opts.Page.W, Ht: opts.Page.H},
|
||||
})
|
||||
|
||||
pdf.SetMargins(opts.Margin, opts.Margin, opts.Margin)
|
||||
pdf.SetAutoPageBreak(true, opts.Margin+footerReserve)
|
||||
|
||||
// Embeds a subset of the TrueType data, which is what makes the Vietnamese
|
||||
// diacritics render instead of falling back to "?". The bytes are passed
|
||||
// directly rather than by path: the path-taking variant joins the name onto
|
||||
// fpdf's own font directory (default "."), which mangles an absolute path
|
||||
// into a working-directory-relative one.
|
||||
pdf.AddUTF8FontFromBytes(bodyFont, "", opts.Font.Data)
|
||||
pdf.SetFont(bodyFont, "", opts.FontSize)
|
||||
pdf.SetTitle(opts.Title, true)
|
||||
|
||||
lineHeight := opts.FontSize * opts.LineSpacing * pointsToMM
|
||||
paragraphGap := lineHeight * 0.45
|
||||
headingSize := opts.FontSize * 1.35
|
||||
|
||||
addFooter(pdf, opts)
|
||||
writeTitlePage(pdf, opts, len(chapters))
|
||||
|
||||
for _, ch := range chapters {
|
||||
pdf.AddPage()
|
||||
|
||||
pdf.SetFontSize(headingSize)
|
||||
pdf.MultiCell(0, headingSize*1.3*pointsToMM, ch.Heading, "", "L", false)
|
||||
pdf.Ln(paragraphGap * 1.6)
|
||||
|
||||
pdf.SetFontSize(opts.FontSize)
|
||||
for _, p := range ch.Paragraphs {
|
||||
// "J" justifies, which keeps the short measure of a phone page tidy.
|
||||
pdf.MultiCell(0, lineHeight, p, "", "J", false)
|
||||
pdf.Ln(paragraphGap)
|
||||
}
|
||||
}
|
||||
|
||||
if err := pdf.OutputFileAndClose(path); err != nil {
|
||||
return fmt.Errorf("write pdf %s: %w", path, err)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// addFooter prints a centred page number, restoring the body font size so the
|
||||
// footer callback cannot leak its own size into the following content.
|
||||
func addFooter(pdf *fpdf.Fpdf, opts Options) {
|
||||
pdf.SetFooterFunc(func() {
|
||||
if pdf.PageNo() <= 1 {
|
||||
return
|
||||
}
|
||||
pdf.SetY(-(opts.Margin + footerReserve*0.6))
|
||||
pdf.SetFontSize(opts.FontSize * 0.75)
|
||||
pdf.SetTextColor(120, 120, 120)
|
||||
pdf.CellFormat(0, 4, fmt.Sprintf("%d", pdf.PageNo()-1), "", 0, "C", false, 0, "")
|
||||
pdf.SetTextColor(0, 0, 0)
|
||||
pdf.SetFontSize(opts.FontSize)
|
||||
})
|
||||
}
|
||||
|
||||
func writeTitlePage(pdf *fpdf.Fpdf, opts Options, chapterCount int) {
|
||||
pdf.AddPage()
|
||||
pdf.SetY(opts.Page.H * 0.30)
|
||||
|
||||
titleSize := opts.FontSize * 1.9
|
||||
pdf.SetFontSize(titleSize)
|
||||
pdf.MultiCell(0, titleSize*1.35*pointsToMM, strings.ToUpper(opts.Title), "", "C", false)
|
||||
|
||||
pdf.Ln(opts.FontSize * pointsToMM * 2)
|
||||
pdf.SetFontSize(opts.FontSize * 0.85)
|
||||
pdf.SetTextColor(90, 90, 90)
|
||||
pdf.MultiCell(0, opts.FontSize*1.4*pointsToMM,
|
||||
fmt.Sprintf("%d chương", chapterCount), "", "C", false)
|
||||
if opts.SourceURL != "" {
|
||||
pdf.MultiCell(0, opts.FontSize*1.4*pointsToMM, opts.SourceURL, "", "C", false)
|
||||
}
|
||||
pdf.SetTextColor(0, 0, 0)
|
||||
pdf.SetFontSize(opts.FontSize)
|
||||
}
|
||||
@@ -0,0 +1,79 @@
|
||||
package pdf
|
||||
|
||||
import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
"testing"
|
||||
)
|
||||
|
||||
func testOptions(t *testing.T) Options {
|
||||
t.Helper()
|
||||
return Options{
|
||||
Page: PhonePage,
|
||||
Margin: 6,
|
||||
Font: BundledFont(),
|
||||
FontSize: 12,
|
||||
LineSpacing: 1.55,
|
||||
Title: "TRỞ LẠI NĂM THÁNG CŨ",
|
||||
SourceURL: "https://example.test/n.html",
|
||||
}
|
||||
}
|
||||
|
||||
func TestWriteProducesReadablePDF(t *testing.T) {
|
||||
path := filepath.Join(t.TempDir(), "out.pdf")
|
||||
|
||||
chapters := []Chapter{
|
||||
{Heading: "Chương 1", Paragraphs: []string{
|
||||
"Nghe vị trưởng tử nói chuyện với nàng.",
|
||||
"Một đoạn văn khác để kiểm tra ngắt dòng tự động trên trang nhỏ.",
|
||||
}},
|
||||
{Heading: "Chương 2", Paragraphs: []string{"Đoạn cuối."}},
|
||||
}
|
||||
|
||||
if err := Write(path, testOptions(t), chapters); err != nil {
|
||||
t.Fatalf("Write: %v", err)
|
||||
}
|
||||
|
||||
info, err := os.Stat(path)
|
||||
if err != nil {
|
||||
t.Fatalf("stat output: %v", err)
|
||||
}
|
||||
if info.Size() == 0 {
|
||||
t.Fatal("wrote an empty PDF")
|
||||
}
|
||||
|
||||
header := make([]byte, 5)
|
||||
f, err := os.Open(path)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
defer f.Close()
|
||||
if _, err := f.Read(header); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if string(header) != "%PDF-" {
|
||||
t.Errorf("output does not start with a PDF header, got %q", header)
|
||||
}
|
||||
}
|
||||
|
||||
// A novel with many chapters must not overflow a page; auto page break plus the
|
||||
// footer reserve handles that, so a long chapter should span several pages.
|
||||
func TestWriteHandlesLongChapters(t *testing.T) {
|
||||
path := filepath.Join(t.TempDir(), "long.pdf")
|
||||
|
||||
paragraphs := make([]string, 200)
|
||||
for i := range paragraphs {
|
||||
paragraphs[i] = "Một đoạn văn dài để buộc trình kết xuất phải sang trang mới nhiều lần."
|
||||
}
|
||||
|
||||
if err := Write(path, testOptions(t), []Chapter{{Heading: "Chương 1", Paragraphs: paragraphs}}); err != nil {
|
||||
t.Fatalf("Write: %v", err)
|
||||
}
|
||||
info, err := os.Stat(path)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if info.Size() < 2000 {
|
||||
t.Errorf("output suspiciously small (%d bytes) for 200 paragraphs", info.Size())
|
||||
}
|
||||
}
|
||||
@@ -12,8 +12,8 @@ import (
|
||||
"github.com/go-telegram/bot"
|
||||
"github.com/go-telegram/bot/models"
|
||||
|
||||
"github.com/tiennm99/monkeyd-crawler/export"
|
||||
crawler "github.com/tiennm99/monkeyd-crawler/monkeyd"
|
||||
"github.com/tiennm99/miti99bot/internal/modules/monkeyd/crawler"
|
||||
"github.com/tiennm99/miti99bot/internal/modules/monkeyd/export"
|
||||
|
||||
"github.com/tiennm99/miti99bot/internal/log"
|
||||
"github.com/tiennm99/miti99bot/internal/modules"
|
||||
|
||||
Vendored
-1
Submodule third_party/monkeyd-crawler deleted from d88f2a482d.
Reference in new issue
Block a user