Files
thptqg/assembler/internal/registry/registry.go
T
tiennm99 c359a0b444 refactor: one directory per pipeline stage, and an assembler to drive them
The repository now reads as the pipeline it is: crawler fetches, parser
converts, assembler verifies and publishes, with data/ and web/ as the stores
they hand work through. go-parser is renamed parser now that there is no other.

The assembler replaces build-db.js and assemble-site.js. It compiles the
parser, builds and verifies each database, compresses it, runs the Vite build
and assembles _site — one command, and the only place that knows the order.

It also closes a real hole: nothing previously asserted that a database reached
the site. An empty staging directory assembled happily, so every page rendered,
every query 404d and CI stayed green. The row-count and size guards could not
catch that, since they only run when a database was built at all.

Removing Node from the root forced the dataset list out of web/src/datasets.js,
which the assembler cannot import. datasets.json is now the registry both sides
read — JSON because Go and the browser both parse it without a dependency —
while presentation stays in the web app, keyed by id and cross-checked against
the registry so a half-added dataset fails instead of half-working.

Guards verified by making each one fail: a missing database, and an expected
row count one higher than the truth.
2026-08-13 22:50:05 +07:00

81 lines
2.2 KiB
Go

// Package registry reads the repository-root datasets.json.
//
// That file is the one place every stage agrees on what exists. The Vite app
// reads it too, which is why it is JSON: Go and the browser both parse it
// without a dependency.
package registry
import (
"encoding/json"
"fmt"
"os"
"path/filepath"
)
// Dataset is one entry in the registry.
type Dataset struct {
ID string `json:"id"`
// ExpectedRows is exact. The inputs are frozen historical exam results, so
// a deviation of even one row means something changed that nobody intended.
ExpectedRows int64 `json:"expectedRows"`
// DbSizeMb is the usual size of the gzipped database, used to catch a
// build that produced a plausible row count but a truncated artifact.
DbSizeMb float64 `json:"dbSizeMb"`
}
type file struct {
Datasets []Dataset `json:"datasets"`
}
// Load reads datasets.json from the repository root.
func Load(root string) ([]Dataset, error) {
path := filepath.Join(root, "datasets.json")
b, err := os.ReadFile(path)
if err != nil {
return nil, fmt.Errorf("cannot read the dataset registry: %w", err)
}
var f file
if err := json.Unmarshal(b, &f); err != nil {
return nil, fmt.Errorf("%s: %w", path, err)
}
if len(f.Datasets) == 0 {
return nil, fmt.Errorf("%s declares no datasets", path)
}
for _, d := range f.Datasets {
switch {
case d.ID == "":
return nil, fmt.Errorf("%s: a dataset has no id", path)
case d.ExpectedRows <= 0:
return nil, fmt.Errorf("%s: %s has no expectedRows; the build guard needs it", path, d.ID)
case d.DbSizeMb <= 0:
return nil, fmt.Errorf("%s: %s has no dbSizeMb; the size guard needs it", path, d.ID)
}
}
return f.Datasets, nil
}
// Select returns the named datasets, or all of them when none are named.
func Select(all []Dataset, ids []string) ([]Dataset, error) {
if len(ids) == 0 {
return all, nil
}
byID := make(map[string]Dataset, len(all))
known := make([]string, 0, len(all))
for _, d := range all {
byID[d.ID] = d
known = append(known, d.ID)
}
out := make([]Dataset, 0, len(ids))
for _, id := range ids {
d, ok := byID[id]
if !ok {
return nil, fmt.Errorf("unknown dataset %q (known: %v)", id, known)
}
out = append(out, d)
}
return out, nil
}