mirror of
https://github.com/tiennm99/thptqg.git
synced 2026-10-11 03:13:48 +00:00
The repository now reads as the pipeline it is: crawler fetches, parser converts, assembler verifies and publishes, with data/ and web/ as the stores they hand work through. go-parser is renamed parser now that there is no other. The assembler replaces build-db.js and assemble-site.js. It compiles the parser, builds and verifies each database, compresses it, runs the Vite build and assembles _site — one command, and the only place that knows the order. It also closes a real hole: nothing previously asserted that a database reached the site. An empty staging directory assembled happily, so every page rendered, every query 404d and CI stayed green. The row-count and size guards could not catch that, since they only run when a database was built at all. Removing Node from the root forced the dataset list out of web/src/datasets.js, which the assembler cannot import. datasets.json is now the registry both sides read — JSON because Go and the browser both parse it without a dependency — while presentation stays in the web app, keyed by id and cross-checked against the registry so a half-added dataset fails instead of half-working. Guards verified by making each one fail: a missing database, and an expected row count one higher than the truth.
154 lines
4.2 KiB
Go
154 lines
4.2 KiB
Go
// Command crawl downloads a dataset's source spreadsheets into data/<id>/.
|
|
//
|
|
// The argument is the dataset id, the same one used by the parser's configs and
|
|
// the published site paths.
|
|
//
|
|
// crawl 2016 # 119 exam-cluster files
|
|
// crawl 2017 # 63 province files
|
|
// crawl 2017 --list # show what would be downloaded
|
|
//
|
|
// Each run reads the download links out of the article that published them, so
|
|
// --list needs network access too. Runs are idempotent: a file already present
|
|
// and non-empty is skipped, so an interrupted crawl can simply be re-run.
|
|
package main
|
|
|
|
import (
|
|
"context"
|
|
"flag"
|
|
"fmt"
|
|
"net/http"
|
|
"os"
|
|
"os/signal"
|
|
"path/filepath"
|
|
"syscall"
|
|
"time"
|
|
|
|
"github.com/tiennm99/thptqg/crawler/internal/article"
|
|
"github.com/tiennm99/thptqg/crawler/internal/fetch"
|
|
"github.com/tiennm99/thptqg/crawler/internal/sources"
|
|
)
|
|
|
|
func main() {
|
|
if err := run(os.Args[1:]); err != nil {
|
|
fmt.Fprintf(os.Stderr, "crawl: %v\n", err)
|
|
os.Exit(1)
|
|
}
|
|
}
|
|
|
|
func usage() {
|
|
fmt.Fprint(os.Stderr, "usage: crawl <dataset> [flags]\n\nDatasets:\n")
|
|
for _, s := range sources.All() {
|
|
fmt.Fprintf(os.Stderr, " %-6s %s\n", s.ID, s.Summary)
|
|
}
|
|
fmt.Fprint(os.Stderr, "\nFlags:\n")
|
|
fmt.Fprint(os.Stderr, " --out string output directory (default ../data/<dataset>)\n")
|
|
fmt.Fprint(os.Stderr, " --concurrency int parallel downloads (default 6)\n")
|
|
fmt.Fprint(os.Stderr, " --timeout duration per-file timeout (default 2m)\n")
|
|
fmt.Fprint(os.Stderr, " --list print the file list and exit\n")
|
|
}
|
|
|
|
func run(args []string) error {
|
|
if len(args) == 0 || args[0] == "-h" || args[0] == "--help" {
|
|
usage()
|
|
if len(args) == 0 {
|
|
return fmt.Errorf("no dataset given")
|
|
}
|
|
return nil
|
|
}
|
|
|
|
src, err := sources.Lookup(args[0])
|
|
if err != nil {
|
|
usage()
|
|
return err
|
|
}
|
|
|
|
fs := flag.NewFlagSet(src.ID, flag.ContinueOnError)
|
|
// The default is relative to the crawler module directory, which is where
|
|
// both `go -C crawler run ./cmd/crawl` and a manual `cd crawler` land.
|
|
out := fs.String("out", filepath.Join("..", "data", src.ID), "output directory")
|
|
concurrency := fs.Int("concurrency", 6, "parallel downloads")
|
|
timeout := fs.Duration("timeout", 2*time.Minute, "per-file timeout")
|
|
list := fs.Bool("list", false, "print the file list and exit")
|
|
if err := fs.Parse(args[1:]); err != nil {
|
|
return err
|
|
}
|
|
|
|
// Ctrl-C cancels the article fetch and any in-flight download; each worker
|
|
// deletes its partial file on the way out, so an interrupted run leaves
|
|
// nothing half-written.
|
|
ctx, stop := signal.NotifyContext(context.Background(), os.Interrupt, syscall.SIGTERM)
|
|
defer stop()
|
|
|
|
client := &http.Client{Timeout: *timeout}
|
|
|
|
fmt.Printf("Reading %s\n", src.Article)
|
|
links, err := article.Fetch(ctx, client, src.Article, src.Headers, src.Exts...)
|
|
if err != nil {
|
|
return err
|
|
}
|
|
files, err := src.Resolve(links)
|
|
if err != nil {
|
|
return err
|
|
}
|
|
|
|
outDir, err := filepath.Abs(*out)
|
|
if err != nil {
|
|
return err
|
|
}
|
|
|
|
items := make([]fetch.Item, 0, len(files))
|
|
for _, f := range files {
|
|
items = append(items, fetch.Item{
|
|
Name: f.Name,
|
|
URL: f.URL,
|
|
Path: filepath.Join(outDir, f.Dest),
|
|
})
|
|
}
|
|
|
|
if *list {
|
|
for _, it := range items {
|
|
fmt.Printf("%s\t%s\n", filepath.Base(it.Path), it.URL)
|
|
}
|
|
return nil
|
|
}
|
|
|
|
fmt.Printf("Downloading %d files to %s...\n", len(items), outDir)
|
|
|
|
results, runErr := fetch.Run(ctx, items, fetch.Options{
|
|
Concurrency: *concurrency,
|
|
Timeout: *timeout,
|
|
Headers: src.Headers,
|
|
OnResult: printResult,
|
|
})
|
|
if runErr != nil {
|
|
return runErr
|
|
}
|
|
|
|
ok, skip, failed := fetch.Tally(results)
|
|
fmt.Printf("\nDone. ok=%d skip=%d fail=%d\n", ok, skip, len(failed))
|
|
if len(failed) > 0 {
|
|
for _, r := range failed {
|
|
fmt.Fprintf(os.Stderr, " %s: %v\n", r.Item.Name, r.Err)
|
|
}
|
|
return fmt.Errorf("%d file(s) failed", len(failed))
|
|
}
|
|
return nil
|
|
}
|
|
|
|
func printResult(done, total int, r fetch.Result) {
|
|
tag := map[fetch.Status]string{
|
|
fetch.StatusOK: "✓",
|
|
fetch.StatusSkip: "·",
|
|
fetch.StatusFail: "✗",
|
|
}[r.Status]
|
|
|
|
detail := ""
|
|
switch r.Status {
|
|
case fetch.StatusFail:
|
|
detail = r.Err.Error()
|
|
default:
|
|
detail = fmt.Sprintf("%.0f KB", float64(r.Size)/1024)
|
|
}
|
|
fmt.Printf(" %s [%d/%d] %-20s %s\n", tag, done, total, r.Item.Name, detail)
|
|
}
|