Files
thptqg/crawler/cmd/crawl/main.go
T
tiennm99 c359a0b444 refactor: one directory per pipeline stage, and an assembler to drive them
The repository now reads as the pipeline it is: crawler fetches, parser
converts, assembler verifies and publishes, with data/ and web/ as the stores
they hand work through. go-parser is renamed parser now that there is no other.

The assembler replaces build-db.js and assemble-site.js. It compiles the
parser, builds and verifies each database, compresses it, runs the Vite build
and assembles _site — one command, and the only place that knows the order.

It also closes a real hole: nothing previously asserted that a database reached
the site. An empty staging directory assembled happily, so every page rendered,
every query 404d and CI stayed green. The row-count and size guards could not
catch that, since they only run when a database was built at all.

Removing Node from the root forced the dataset list out of web/src/datasets.js,
which the assembler cannot import. datasets.json is now the registry both sides
read — JSON because Go and the browser both parse it without a dependency —
while presentation stays in the web app, keyed by id and cross-checked against
the registry so a half-added dataset fails instead of half-working.

Guards verified by making each one fail: a missing database, and an expected
row count one higher than the truth.
2026-08-13 22:50:05 +07:00

154 lines
4.2 KiB
Go

// Command crawl downloads a dataset's source spreadsheets into data/<id>/.
//
// The argument is the dataset id, the same one used by the parser's configs and
// the published site paths.
//
// crawl 2016 # 119 exam-cluster files
// crawl 2017 # 63 province files
// crawl 2017 --list # show what would be downloaded
//
// Each run reads the download links out of the article that published them, so
// --list needs network access too. Runs are idempotent: a file already present
// and non-empty is skipped, so an interrupted crawl can simply be re-run.
package main
import (
"context"
"flag"
"fmt"
"net/http"
"os"
"os/signal"
"path/filepath"
"syscall"
"time"
"github.com/tiennm99/thptqg/crawler/internal/article"
"github.com/tiennm99/thptqg/crawler/internal/fetch"
"github.com/tiennm99/thptqg/crawler/internal/sources"
)
func main() {
if err := run(os.Args[1:]); err != nil {
fmt.Fprintf(os.Stderr, "crawl: %v\n", err)
os.Exit(1)
}
}
func usage() {
fmt.Fprint(os.Stderr, "usage: crawl <dataset> [flags]\n\nDatasets:\n")
for _, s := range sources.All() {
fmt.Fprintf(os.Stderr, " %-6s %s\n", s.ID, s.Summary)
}
fmt.Fprint(os.Stderr, "\nFlags:\n")
fmt.Fprint(os.Stderr, " --out string output directory (default ../data/<dataset>)\n")
fmt.Fprint(os.Stderr, " --concurrency int parallel downloads (default 6)\n")
fmt.Fprint(os.Stderr, " --timeout duration per-file timeout (default 2m)\n")
fmt.Fprint(os.Stderr, " --list print the file list and exit\n")
}
func run(args []string) error {
if len(args) == 0 || args[0] == "-h" || args[0] == "--help" {
usage()
if len(args) == 0 {
return fmt.Errorf("no dataset given")
}
return nil
}
src, err := sources.Lookup(args[0])
if err != nil {
usage()
return err
}
fs := flag.NewFlagSet(src.ID, flag.ContinueOnError)
// The default is relative to the crawler module directory, which is where
// both `go -C crawler run ./cmd/crawl` and a manual `cd crawler` land.
out := fs.String("out", filepath.Join("..", "data", src.ID), "output directory")
concurrency := fs.Int("concurrency", 6, "parallel downloads")
timeout := fs.Duration("timeout", 2*time.Minute, "per-file timeout")
list := fs.Bool("list", false, "print the file list and exit")
if err := fs.Parse(args[1:]); err != nil {
return err
}
// Ctrl-C cancels the article fetch and any in-flight download; each worker
// deletes its partial file on the way out, so an interrupted run leaves
// nothing half-written.
ctx, stop := signal.NotifyContext(context.Background(), os.Interrupt, syscall.SIGTERM)
defer stop()
client := &http.Client{Timeout: *timeout}
fmt.Printf("Reading %s\n", src.Article)
links, err := article.Fetch(ctx, client, src.Article, src.Headers, src.Exts...)
if err != nil {
return err
}
files, err := src.Resolve(links)
if err != nil {
return err
}
outDir, err := filepath.Abs(*out)
if err != nil {
return err
}
items := make([]fetch.Item, 0, len(files))
for _, f := range files {
items = append(items, fetch.Item{
Name: f.Name,
URL: f.URL,
Path: filepath.Join(outDir, f.Dest),
})
}
if *list {
for _, it := range items {
fmt.Printf("%s\t%s\n", filepath.Base(it.Path), it.URL)
}
return nil
}
fmt.Printf("Downloading %d files to %s...\n", len(items), outDir)
results, runErr := fetch.Run(ctx, items, fetch.Options{
Concurrency: *concurrency,
Timeout: *timeout,
Headers: src.Headers,
OnResult: printResult,
})
if runErr != nil {
return runErr
}
ok, skip, failed := fetch.Tally(results)
fmt.Printf("\nDone. ok=%d skip=%d fail=%d\n", ok, skip, len(failed))
if len(failed) > 0 {
for _, r := range failed {
fmt.Fprintf(os.Stderr, " %s: %v\n", r.Item.Name, r.Err)
}
return fmt.Errorf("%d file(s) failed", len(failed))
}
return nil
}
func printResult(done, total int, r fetch.Result) {
tag := map[fetch.Status]string{
fetch.StatusOK: "✓",
fetch.StatusSkip: "·",
fetch.StatusFail: "✗",
}[r.Status]
detail := ""
switch r.Status {
case fetch.StatusFail:
detail = r.Err.Error()
default:
detail = fmt.Sprintf("%.0f KB", float64(r.Size)/1024)
}
fmt.Printf(" %s [%d/%d] %-20s %s\n", tag, done, total, r.Item.Name, detail)
}