Merge pull request #7 from tiennm99/chore/remove-js-and-dead-code

chore: remove the last JS script and the dead weight three audits found
This commit is contained in:
tiennm99 authored and GitHub committed 2026-08-14 00:08:22 +07:00
commit 3995e9d864
33 files changed
+749 -8814

No files matched your search

+6
View File
@@ -62,6 +62,12 @@ jobs:
- name: Test crawler
run: go -C crawler test ./...
# The assembler owns every guard between a built database and the
# published site, so its tests are the ones that prove a short or missing
# database cannot ship.
- name: Test assembler
run: go -C assembler test ./...
- name: Lint web
working-directory: web
run: npm run lint
+1
View File
@@ -59,6 +59,7 @@ is missing altogether. Sub-steps when iterating:
```bash
go -C assembler run ./cmd/assemble db 2017 # one database
go -C assembler run ./cmd/assemble site # web build and _site only
go -C assembler run ./cmd/assemble verify A B # compare two sets of databases
(cd web && npm run dev) # the app against staged databases
```
+69 -1
View File
@@ -4,6 +4,7 @@
// assemble # databases, then the site
// assemble db # databases only (add ids to limit: assemble db 2017)
// assemble site # web build and _site only, reusing staged databases
// assemble verify A B # compare two sets of built databases field by field
//
// It sequences the other stages rather than doing their work: the parser reads
// spreadsheets, Vite bundles the app, and this decides what runs, checks what
@@ -18,6 +19,7 @@ import (
"github.com/tiennm99/thptqg/assembler/internal/databases"
"github.com/tiennm99/thptqg/assembler/internal/registry"
"github.com/tiennm99/thptqg/assembler/internal/site"
"github.com/tiennm99/thptqg/assembler/internal/verify"
)
func main() {
@@ -34,9 +36,20 @@ Steps:
(none) databases, then the site
db build, verify and compress the databases
site build the web app and assemble _site
verify A B compare two directories of built databases, field by field
Naming datasets limits the db step to those; the site step always covers all of
them, since a partial site would publish links to databases it did not build.
verify takes two directories holding <id>.db.gz (or <id>.db) — for example a
copy of the previous .build/public/db against a fresh build. Relative paths
resolve against the repository root, not the working directory. It reports row
counts, per-column non-NULL counts, a full-table hash and the first differing
rows.
Nothing else checks database content: the reader oracle covers reading and the
row-count guard covers how many rows, but a change in transform or writer logic
can alter what is in them while both still pass.
`)
}
@@ -44,7 +57,7 @@ func run(args []string) error {
step := ""
if len(args) > 0 {
switch args[0] {
case "db", "site":
case "db", "site", "verify":
step, args = args[0], args[1:]
case "-h", "--help":
usage()
@@ -66,6 +79,19 @@ func run(args []string) error {
return err
}
if step == "verify" {
if len(args) != 2 {
usage()
return fmt.Errorf("verify needs two directories")
}
// Relative paths resolve against the repository root, not the working
// directory. The documented invocation is `go -C assembler run ...`,
// and -C moves the process into assembler/ before main starts — so a
// bare `.build/public/db` would otherwise look for it in there and
// report a missing database that is sitting in plain sight.
return runVerify(all, atRoot(root, args[0]), atRoot(root, args[1]))
}
if step == "" || step == "db" {
selected, err := registry.Select(all, args)
if err != nil {
@@ -88,6 +114,48 @@ func run(args []string) error {
return nil
}
// atRoot resolves a relative path against the repository root.
func atRoot(root, path string) string {
if filepath.IsAbs(path) {
return path
}
return filepath.Join(root, path)
}
// runVerify compares two sets of built databases and fails loudly on any
// difference. A dataset missing from either side is an error rather than a
// skip: silently comparing one of two datasets is how a gate passes without
// proving anything.
func runVerify(all []registry.Dataset, dirA, dirB string) error {
results, err := verify.Compare(all, dirA, dirB)
if err != nil {
return err
}
bad := 0
for _, r := range results {
fmt.Printf("--- %s ---\n", r.ID)
fmt.Printf(" rows : %d vs %d\n", r.RowsA, r.RowsB)
if r.OK() {
fmt.Printf(" full-table : identical sha256 %s\n", r.HashA[:16])
continue
}
bad++
for _, p := range r.Problems {
fmt.Printf(" *** %s\n", p)
}
for _, d := range r.FirstDiff {
fmt.Println(d)
}
}
if bad > 0 {
return fmt.Errorf("%d dataset(s) differ", bad)
}
fmt.Println("\nIdentical — rows, per-column non-NULL counts, schema and full-table hash.")
return nil
}
func buildDatabases(root string, all, selected []registry.Dataset) error {
p := databases.DefaultPaths(root)
-2
View File
@@ -26,7 +26,6 @@ import (
// Paths locates the pieces this package needs.
type Paths struct {
Root string
// Web is the Vite project directory.
Web string
// Dist is where Vite emits, inside the web workspace.
@@ -39,7 +38,6 @@ type Paths struct {
func DefaultPaths(root string) Paths {
web := filepath.Join(root, "web")
return Paths{
Root: root,
Web: web,
Dist: filepath.Join(web, "dist"),
Site: filepath.Join(root, "_site"),
+2 -2
View File
@@ -22,7 +22,7 @@ func fakeBuild(t *testing.T, dbs ...string) Paths {
for _, name := range dbs {
write(t, filepath.Join(dist, "db", name), "gzipped-bytes")
}
return Paths{Root: root, Web: filepath.Join(root, "web"), Dist: dist, Site: filepath.Join(root, "_site")}
return Paths{Web: filepath.Join(root, "web"), Dist: dist, Site: filepath.Join(root, "_site")}
}
func write(t *testing.T, path, body string) {
@@ -113,7 +113,7 @@ func TestGzipIsNotMistakenForRaw(t *testing.T) {
func TestAssembleRejectsAMissingBuild(t *testing.T) {
root := t.TempDir()
p := Paths{Root: root, Web: root, Dist: filepath.Join(root, "dist"), Site: filepath.Join(root, "_site")}
p := Paths{Web: root, Dist: filepath.Join(root, "dist"), Site: filepath.Join(root, "_site")}
if err := Assemble(p, datasets); err == nil {
t.Fatal("expected an error when there is no Vite build")
}
+384
View File
@@ -0,0 +1,384 @@
// Package verify compares two sets of built databases field by field.
//
// Nothing else checks database *content*. The reader-fidelity oracle covers
// reading the spreadsheets, and the row-count guard covers how many rows came
// out — but a change in transform or writer logic can alter what is in those
// rows while both of those still pass. This is what catches that.
//
// It was written as the gate for the Rust-to-Go parser migration and works for
// any two builds: before and after a refactor, or a re-crawl against the
// databases already published.
package verify
import (
"compress/gzip"
"crypto/sha256"
"database/sql"
"encoding/hex"
"fmt"
"io"
"math"
"os"
"path/filepath"
"sort"
"strconv"
"strings"
_ "modernc.org/sqlite"
"github.com/tiennm99/thptqg/assembler/internal/registry"
)
const driverName = "sqlite"
// Result is the outcome for one dataset.
type Result struct {
ID string
Problems []string
RowsA int64
RowsB int64
HashA string
HashB string
FirstDiff []string
}
// OK reports whether the two databases are logically identical.
func (r Result) OK() bool { return len(r.Problems) == 0 }
// Compare checks every dataset in the registry, reading <dir>/<id>.db.gz (or
// <id>.db) from each side.
func Compare(datasets []registry.Dataset, dirA, dirB string) ([]Result, error) {
out := make([]Result, 0, len(datasets))
for _, d := range datasets {
r, err := compareOne(d.ID, dirA, dirB)
if err != nil {
return nil, err
}
out = append(out, r)
}
return out, nil
}
func compareOne(id, dirA, dirB string) (Result, error) {
res := Result{ID: id}
a, cleanA, err := open(dirA, id)
if err != nil {
return res, err
}
defer cleanA()
defer a.Close()
b, cleanB, err := open(dirB, id)
if err != nil {
return res, err
}
defer cleanB()
defer b.Close()
sigA, err := schemaSignature(a)
if err != nil {
return res, err
}
sigB, err := schemaSignature(b)
if err != nil {
return res, err
}
if sigA != sigB {
res.Problems = append(res.Problems, "schema/index metadata differs")
}
cols, err := columns(a)
if err != nil {
return res, err
}
if res.RowsA, err = count(a, "SELECT COUNT(*) FROM student"); err != nil {
return res, err
}
if res.RowsB, err = count(b, "SELECT COUNT(*) FROM student"); err != nil {
return res, err
}
if res.RowsA != res.RowsB {
res.Problems = append(res.Problems, fmt.Sprintf("row count %d vs %d", res.RowsA, res.RowsB))
}
// Per-column non-NULL counts localise a difference to a column even when the
// full-table hash has already said "something differs".
for _, c := range cols {
q := fmt.Sprintf("SELECT COUNT(%s) FROM student", c)
na, err := count(a, q)
if err != nil {
return res, err
}
nb, err := count(b, q)
if err != nil {
return res, err
}
if na != nb {
res.Problems = append(res.Problems, fmt.Sprintf("non-NULL count for %s: %d vs %d", c, na, nb))
}
}
if res.HashA, err = tableHash(a, cols); err != nil {
return res, err
}
if res.HashB, err = tableHash(b, cols); err != nil {
return res, err
}
if res.HashA != res.HashB {
res.Problems = append(res.Problems, "full-table hash differs")
if res.FirstDiff, err = firstDifferences(a, b, cols, 20); err != nil {
return res, err
}
}
return res, nil
}
// open finds <id>.db.gz or <id>.db in dir and returns a handle. A compressed
// database is expanded to a temporary file, since SQLite needs to seek.
func open(dir, id string) (*sql.DB, func(), error) {
noop := func() {}
plain := filepath.Join(dir, id+".db")
if _, err := os.Stat(plain); err == nil {
db, err := sql.Open(driverName, "file:"+plain+"?mode=ro")
return db, noop, err
}
gzPath := filepath.Join(dir, id+".db.gz")
f, err := os.Open(gzPath)
if err != nil {
return nil, noop, fmt.Errorf("no %s.db or %s.db.gz in %s", id, id, dir)
}
defer f.Close()
zr, err := gzip.NewReader(f)
if err != nil {
return nil, noop, fmt.Errorf("%s: %w", gzPath, err)
}
defer zr.Close()
tmp, err := os.CreateTemp("", "verify-"+id+"-*.db")
if err != nil {
return nil, noop, err
}
cleanup := func() { os.Remove(tmp.Name()) }
if _, err := io.Copy(tmp, zr); err != nil {
tmp.Close()
cleanup()
return nil, noop, err
}
if err := tmp.Close(); err != nil {
cleanup()
return nil, noop, err
}
db, err := sql.Open(driverName, "file:"+tmp.Name()+"?mode=ro")
if err != nil {
cleanup()
return nil, noop, err
}
return db, cleanup, nil
}
func count(db *sql.DB, query string) (int64, error) {
var n int64
err := db.QueryRow(query).Scan(&n)
return n, err
}
// columns returns the column names in declaration order.
func columns(db *sql.DB) ([]string, error) {
rows, err := db.Query("SELECT name FROM pragma_table_info('student') ORDER BY cid")
if err != nil {
return nil, err
}
defer rows.Close()
var out []string
for rows.Next() {
var name string
if err := rows.Scan(&name); err != nil {
return nil, err
}
out = append(out, name)
}
if err := rows.Err(); err != nil {
return nil, err
}
if len(out) == 0 {
return nil, fmt.Errorf("no student table")
}
return out, nil
}
// schemaSignature reduces the table and index metadata to a comparable string.
func schemaSignature(db *sql.DB) (string, error) {
rows, err := db.Query("SELECT cid, name, type, \"notnull\", pk FROM pragma_table_info('student') ORDER BY cid")
if err != nil {
return "", err
}
var cols []string
for rows.Next() {
var cid, notnull, pk int
var name, typ string
if err := rows.Scan(&cid, &name, &typ, &notnull, &pk); err != nil {
rows.Close()
return "", err
}
cols = append(cols, fmt.Sprintf("%d:%s:%s:%d:%d", cid, name, typ, notnull, pk))
}
rows.Close()
if err := rows.Err(); err != nil {
return "", err
}
idxRows, err := db.Query("SELECT name, \"unique\", partial FROM pragma_index_list('student')")
if err != nil {
return "", err
}
defer idxRows.Close()
var idx []string
for idxRows.Next() {
var name string
var uniq, partial int
if err := idxRows.Scan(&name, &uniq, &partial); err != nil {
return "", err
}
idx = append(idx, fmt.Sprintf("%s:%d:%d", name, uniq, partial))
}
if err := idxRows.Err(); err != nil {
return "", err
}
// Index order is not guaranteed by SQLite; sort so it cannot cause a false
// difference.
sort.Strings(idx)
return strings.Join(cols, "|") + "\n" + strings.Join(idx, "|"), nil
}
// Field and record separators. The values are hashed with explicit delimiters
// so that ("ab","c") and ("a","bc") cannot produce the same digest — joining
// them bare would make a shifted field boundary invisible.
const (
fieldSep = "\x1f"
recordSep = "\x1e"
)
// ser renders one value canonically.
//
// NULL gets a sentinel no real value can collide with, and a whole-numbered
// REAL is rendered with one decimal place so that a column holding 3 and one
// holding 3.0 compare equal, as SQLite considers them.
func ser(v any) string {
switch t := v.(type) {
case nil:
return "\x00NULL"
case int64:
return strconv.FormatInt(t, 10)
case float64:
if t == math.Trunc(t) && !math.IsInf(t, 0) {
return strconv.FormatFloat(t, 'f', 1, 64)
}
return strconv.FormatFloat(t, 'g', -1, 64)
case []byte:
return string(t)
case string:
return t
default:
return fmt.Sprint(t)
}
}
// tableHash is a rolling SHA-256 over every row, ordered by primary key.
//
// Streamed rather than materialised: 877k rows by 22 columns would otherwise be
// a large amount of memory for no benefit.
func tableHash(db *sql.DB, cols []string) (string, error) {
rows, err := db.Query("SELECT * FROM student ORDER BY so_bao_danh")
if err != nil {
return "", err
}
defer rows.Close()
h := sha256.New()
vals := make([]any, len(cols))
ptrs := make([]any, len(cols))
for i := range vals {
ptrs[i] = &vals[i]
}
for rows.Next() {
if err := rows.Scan(ptrs...); err != nil {
return "", err
}
for i := range vals {
io.WriteString(h, ser(vals[i]))
io.WriteString(h, fieldSep)
}
io.WriteString(h, recordSep)
}
if err := rows.Err(); err != nil {
return "", err
}
return hex.EncodeToString(h.Sum(nil)), nil
}
// firstDifferences walks both tables in step and reports where they diverge,
// so a failure names a row and a column rather than only a digest.
func firstDifferences(a, b *sql.DB, cols []string, limit int) ([]string, error) {
ra, err := a.Query("SELECT * FROM student ORDER BY so_bao_danh")
if err != nil {
return nil, err
}
defer ra.Close()
rb, err := b.Query("SELECT * FROM student ORDER BY so_bao_danh")
if err != nil {
return nil, err
}
defer rb.Close()
scan := func(rows *sql.Rows) ([]string, bool, error) {
if !rows.Next() {
return nil, false, rows.Err()
}
vals := make([]any, len(cols))
ptrs := make([]any, len(cols))
for i := range vals {
ptrs[i] = &vals[i]
}
if err := rows.Scan(ptrs...); err != nil {
return nil, false, err
}
out := make([]string, len(cols))
for i := range vals {
out[i] = ser(vals[i])
}
return out, true, nil
}
var out []string
for len(out) < limit {
va, okA, err := scan(ra)
if err != nil {
return nil, err
}
vb, okB, err := scan(rb)
if err != nil {
return nil, err
}
if !okA || !okB {
break
}
for i, c := range cols {
if va[i] != vb[i] {
out = append(out, fmt.Sprintf(" so_bao_danh=%s %s: a=%q b=%q", va[0], c, va[i], vb[i]))
break
}
}
}
return out, nil
}
+165
View File
@@ -0,0 +1,165 @@
package verify
import (
"database/sql"
"os"
"path/filepath"
"strings"
"testing"
"github.com/tiennm99/thptqg/assembler/internal/registry"
)
// makeDB writes a miniature student table so the comparison logic can be
// exercised without building a real 877k-row database.
func makeDB(t *testing.T, dir, id string, rows [][3]any) {
t.Helper()
path := filepath.Join(dir, id+".db")
if err := os.MkdirAll(dir, 0o755); err != nil {
t.Fatal(err)
}
db, err := sql.Open(driverName, path)
if err != nil {
t.Fatal(err)
}
defer db.Close()
if _, err := db.Exec(`CREATE TABLE student (
so_bao_danh TEXT PRIMARY KEY,
ho_ten TEXT NOT NULL,
toan REAL
); CREATE INDEX idx_ho_ten ON student(ho_ten);`); err != nil {
t.Fatal(err)
}
for _, r := range rows {
if _, err := db.Exec("INSERT INTO student VALUES (?,?,?)", r[0], r[1], r[2]); err != nil {
t.Fatal(err)
}
}
}
var base = [][3]any{
{"001", "Nguyễn Văn A", 8.0},
{"002", "Trần Thị B", 7.25},
{"003", "Lê Văn C", nil},
}
func compare(t *testing.T, a, b [][3]any) Result {
t.Helper()
dirA, dirB := t.TempDir(), t.TempDir()
makeDB(t, dirA, "x", a)
makeDB(t, dirB, "x", b)
res, err := Compare([]registry.Dataset{{ID: "x"}}, dirA, dirB)
if err != nil {
t.Fatal(err)
}
return res[0]
}
func TestIdenticalDatabasesMatch(t *testing.T) {
got := compare(t, base, base)
if !got.OK() {
t.Fatalf("identical databases reported problems: %v", got.Problems)
}
if got.HashA != got.HashB || got.HashA == "" {
t.Errorf("hashes should be equal and non-empty: %q %q", got.HashA, got.HashB)
}
if got.RowsA != 3 {
t.Errorf("RowsA = %d, want 3", got.RowsA)
}
}
// TestOneChangedScoreIsCaught is the whole point: a single altered field, with
// the row count and every non-NULL count unchanged, must still fail.
func TestOneChangedScoreIsCaught(t *testing.T) {
changed := [][3]any{{"001", "Nguyễn Văn A", 8.25}, base[1], base[2]}
got := compare(t, base, changed)
if got.OK() {
t.Fatal("a changed score was not detected")
}
if got.RowsA != got.RowsB {
t.Error("row counts should still match — that is what makes this hard")
}
joined := strings.Join(got.Problems, "; ")
if !strings.Contains(joined, "full-table hash differs") {
t.Errorf("problems = %v", got.Problems)
}
// The diagnosis must name the row and the column.
diff := strings.Join(got.FirstDiff, "\n")
if !strings.Contains(diff, "so_bao_danh=001") || !strings.Contains(diff, "toan") {
t.Errorf("FirstDiff should locate the change, got: %v", got.FirstDiff)
}
}
// TestNullVersusValueIsCaught: a value becoming NULL changes the per-column
// count as well, so both signals should fire.
func TestNullVersusValueIsCaught(t *testing.T) {
nulled := [][3]any{base[0], base[1], {"003", "Lê Văn C", 5.0}}
got := compare(t, base, nulled)
if got.OK() {
t.Fatal("a NULL becoming a value was not detected")
}
if !strings.Contains(strings.Join(got.Problems, ";"), "non-NULL count for toan") {
t.Errorf("expected a per-column count difference, got %v", got.Problems)
}
}
func TestRowCountDifferenceIsCaught(t *testing.T) {
got := compare(t, base, base[:2])
if got.OK() {
t.Fatal("a missing row was not detected")
}
if !strings.Contains(strings.Join(got.Problems, ";"), "row count 3 vs 2") {
t.Errorf("problems = %v", got.Problems)
}
}
// TestTextShiftAcrossFieldsIsCaught guards the field separator. Hashing the
// values joined bare would make ("ab","c") and ("a","bc") identical, so a value
// shifted across a column boundary would slip through.
func TestTextShiftAcrossFieldsIsCaught(t *testing.T) {
a := [][3]any{{"001", "ab", 1.0}}
b := [][3]any{{"001", "a", 1.0}}
// Same idea at the boundary between so_bao_danh and ho_ten.
c := [][3]any{{"01", "23", 1.0}}
d := [][3]any{{"012", "3", 1.0}}
if compare(t, a, b).OK() {
t.Error("a shortened text value was not detected")
}
if compare(t, c, d).OK() {
t.Error("a value shifted across a field boundary was not detected")
}
}
func TestMissingDatabaseIsAnError(t *testing.T) {
dir := t.TempDir()
makeDB(t, dir, "x", base)
if _, err := Compare([]registry.Dataset{{ID: "x"}}, dir, t.TempDir()); err == nil {
t.Fatal("comparing against a directory with no database must fail")
}
}
func TestSer(t *testing.T) {
for _, tc := range []struct {
in any
want string
}{
{nil, "\x00NULL"},
{int64(7), "7"},
{8.0, "8.0"}, // whole REALs get a fixed rendering
{7.25, "7.25"},
{"x", "x"},
{[]byte("y"), "y"},
} {
if got := ser(tc.in); got != tc.want {
t.Errorf("ser(%v) = %q, want %q", tc.in, got, tc.want)
}
}
// A NULL must not collide with any real text value.
if ser(nil) == ser("NULL") || ser(nil) == ser("") {
t.Error("the NULL sentinel collides with a real value")
}
}
+1 -1
View File
@@ -52,7 +52,7 @@ func resolveFixture(t *testing.T, src Source) []File {
// parser sorts its inputs and inserts last-wins, so filenames decide which
// row survives a duplicate exam number. If extraction or naming drifted, a
// re-crawl could rebuild a database with the same row count and different
// content, which the row-count guard in build-db.js would not catch.
// content, which the assembler's row-count guard would not catch.
//
// The committed data/<id>/ is the oracle: reading the source article and
// applying the source's naming rule must reproduce it exactly, both directions.
+40 -17
View File
@@ -133,10 +133,22 @@ the 2017 configs listed 14. Neither list was complete: candidates could sit
German, Japanese and Russian in both years, so 1,691 students ended up with **no
foreign-language score at all**.
Running all 16 patterns everywhere recovered them — 182 Russian in 2016, and
German and Japanese across all three 2017 generations. See
`plans/reports/parser-parity-result.md` for the evidence that these are real
scores rather than false matches.
Running all 16 patterns everywhere recovered them: 182 × `tieng_nga` in 2016,
and 93 × `tieng_duc` plus 512 × `tieng_nhat` in 2017.
Four things established that these are real scores and not false regex matches,
checked when the change was made:
- Every student holds zero or exactly one foreign language, never two, so
nothing was double-counted.
- Each affected student had **all** language columns NULL beforehand — e.g. SBD
`01003198` went from all-NULL to `tieng_duc = 8`.
- The counts tracked dataset size across the three 2017 publications that then
existed (93/85/22 German, 512/484/313 Japanese).
- Every recovered value is an ordinary exam score in the 0–10 range.
Row counts and the non-NULL counts of all 18 pre-existing columns were unchanged
across every dataset; only the four language columns gained values.
## Per-dataset quirks
@@ -166,19 +178,26 @@ figure in the table above, and each `.db.gz` must be at least 90% of its usual
size, or the build fails rather than publishing. That guard is the reason a
truncated dataset cannot reach the site with a green pipeline.
For a deeper check, `parser/scripts/differential-parity.mjs` compares two sets
of databases field-by-field — row counts, per-column non-NULL counts, a
full-table SHA-256 over every row ordered by `so_bao_danh`, schema metadata, and
build stdout:
For a deeper check, `assemble verify` compares two sets of built databases
field by field — row counts, per-column non-NULL counts, schema metadata, and a
full-table SHA-256 over every row ordered by `so_bao_danh`:
```bash
node parser/scripts/differential-parity.mjs \
--rust /path/to/a-{id}.db --go /path/to/b-{id}.db
cp -r .build/public/db /tmp/before # keep the databases you have
go -C assembler run ./cmd/assemble db # rebuild
go -C assembler run ./cmd/assemble verify /tmp/before .build/public/db
```
It exits non-zero on any mismatch and fails loudly if a dataset is missing rather
than skipping it. Written for the Rust-to-Go migration, it works for any two
builds. Uses the built-in `node:sqlite`, so it needs no dependencies.
Each side is a directory of `<id>.db.gz` (or `<id>.db`); compressed databases are
expanded to a temporary file automatically. It exits non-zero on any mismatch,
names the first differing rows and columns, and fails rather than skipping when a
dataset is absent from either side — silently comparing one of two datasets is
how a gate passes without proving anything.
This is the only check on database *content*. The reader oracle below covers
reading the spreadsheets and the row-count guard covers how many rows came out,
but a change in transform or writer logic can alter what is in those rows while
both of those still pass.
`parser/internal/reader` additionally carries a frozen oracle of per-file
cell-dump hashes covering all 182 inputs; `go -C parser test ./...` fails if any single
@@ -194,8 +213,8 @@ go -C assembler run ./cmd/assemble db 2017
The row-count guard in `build:db` confirms the rebuild matches the expected
total. That guard checks the count only, so if the crawl was expected to change
the data, compare content with `differential-parity.mjs` against a copy of the
previous database rather than trusting the count.
the data, compare content with `assemble verify` against a copy of the previous
database rather than trusting the count.
## Removed scripts
@@ -203,8 +222,12 @@ previous database rather than trusting the count.
were dropped with the Rust parser. The first two had been broken since before the
repo was unified (a hardcoded Windows path in one, an undeclared
`better-sqlite3` dependency in the other) and neither had any automated caller.
The latter two are superseded by `differential-parity.mjs`, which compares more
and cannot silently skip a dataset.
The latter two are superseded by `assemble verify`, which compares more and
cannot silently skip a dataset. `differential-parity.mjs`, which was that
comparator, has itself been folded into the assembler as
`assembler/internal/verify` — the pipeline is Go outside `web/`, and the port
also fixed a weakness: the JavaScript version hashed each row's fields joined
bare, so a value shifted across a column boundary produced the same digest.
`crawl-baotintuc.js` was not dropped but rewritten as the Go `crawler/` module,
producing the same local filenames. Two changes beyond the port:
+2 -2
View File
@@ -8,7 +8,7 @@ One-time setup: **Settings → Pages → Source: GitHub Actions**.
## What the workflow does
1. Checkout, Go toolchain, Node 24, `npm ci` in `web/`
2. Parser and crawler test suites, web lint, `govulncheck` over all three modules
2. Parser, crawler and assembler test suites, web lint, `govulncheck` over all three modules
3. `go -C assembler run ./cmd/assemble` — the whole pipeline: compile the
parser, build and verify each database, compress it into `.build/public/db/`,
run the Vite build, assemble `_site/`
@@ -85,7 +85,7 @@ artifact — one missing line away from publishing it.
- **No server-side compression is assumed.** The app fetches the `.gz` bytes
directly rather than relying on `Content-Encoding: gzip`; Pages does not
reliably compress arbitrary paths on the fly.
- **Total artifact is about 177 MB**, well inside the 1 GB site limit.
- **Total artifact is about 93 MB**, well inside the 1 GB site limit.
## Rollback
+9 -2
View File
@@ -49,7 +49,7 @@ containing it.
The port was gated on a field-by-field comparison of both implementations across
the four datasets that existed then — 3,265,641 rows with identical full-table
SHA-256, identical per-column non-NULL counts, identical schema metadata and
identical build stdout. `scripts/differential-parity.mjs` is that comparator and
identical build stdout. `assemble verify` is that comparator today and
still runs against any two sets of databases.
Behaviour was matched bug-for-bug, deliberately. Several quirks look like
@@ -70,7 +70,14 @@ Each has a test naming it, so none can be tidied away by accident.
`testdata/reader-fidelity-hashes.tsv` holds a SHA-256 per input file over a
canonical dump of every cell of every sheet. It is **frozen**: it was produced
by the Rust reader, which no longer exists, so it cannot be regenerated. It
still fails if any single cell of any of the 182 files reads differently.
still fails if any single cell of any input file reads differently.
A mismatch names the file but not the cell. `cmd/dumpcells` prints the stream
the hash is taken over, so two runs can be diffed:
```bash
go -C parser run ./cmd/dumpcells ../data/2017/an-giang.xls out.tsv
```
The assembler refuses to publish a database whose row count does not match
the known figure, or whose artifact is under 90% of its usual size.
+8 -3
View File
@@ -1,6 +1,11 @@
// Command dumpcells emits the canonical cell rendering of a spreadsheet, for
// comparison against the Rust/calamine ground truth produced by
// parser/examples/dump_cells.rs.
// Command dumpcells emits the canonical cell rendering of a spreadsheet.
//
// It exists to diagnose a reader-fidelity failure. That suite compares a
// SHA-256 per input file against a frozen oracle, so a mismatch says which file
// changed but not which cell; this prints the stream the hash is taken over, so
// two runs can be diffed. It was originally written to compare against the
// Rust/calamine ground truth from parser/examples/dump_cells.rs, which was
// removed with the Rust tree — the hashes it produced are what remains.
//
// The canonical stream carries geometry and rendered cell values only. The
// calamine Data variant is deliberately excluded: Data::Empty and
+2 -2
View File
@@ -1,8 +1,8 @@
// Command xlsxread reads .xls/.xlsx files and builds SQLite databases for the
// thptqg datasets.
//
// The CLI contract is fixed by parser/scripts/build-db.js and must match the
// Rust binary exactly:
// The CLI contract is fixed by the assembler, which drives this binary, and it
// matched the Rust binary it replaced exactly:
//
// xlsxread build --schema <config.yml> --input <dir> --output <db>
// xlsxread audit --schema <config.yml> --input <dir> --db <db>
+2 -2
View File
@@ -2,8 +2,8 @@
//
// The config deliberately carries no SQL. The table shape, the INSERT and the
// subject regexes are identical for every dataset and live in internal/schema;
// keeping them here meant four copies of the same DDL, which is how the 2016 and
// 2017 schemas drifted apart.
// keeping them here meant one copy of the DDL per dataset, which is how the
// 2016 and 2017 schemas drifted apart.
package config
import (
+1 -1
View File
@@ -309,7 +309,7 @@ func Detect2016(cfg *config.DatasetConfig, inputDir, outputPath string) error {
if err := tx.Commit(); err != nil {
return fmt.Errorf("commit: %w", err)
}
return writer.Finish(db, outputPath, st, label, false)
return writer.Finish(db, outputPath, st)
}
// processFile2016 detects the layout per SHEET and processes that sheet's rows.
+1 -2
View File
@@ -147,7 +147,6 @@ func Standard(cfg *config.DatasetConfig, inputDir, outputPath string) error {
}
defer db.Close()
isOld2 := strings.Contains(label, "old2")
stripBlank := cfg.Reader.StripBlankRows
// One transaction spans the whole dataset directory (main.rs:120,184).
@@ -230,7 +229,7 @@ func Standard(cfg *config.DatasetConfig, inputDir, outputPath string) error {
}
// VACUUM only after COMMIT — SQLite refuses it inside a transaction.
return writer.Finish(db, outputPath, st, label, isOld2)
return writer.Finish(db, outputPath, st)
}
// cellAt returns the trimmed cell at idx, or "" when out of range — the
+1 -1
View File
@@ -30,7 +30,7 @@ import (
// under -short, which is how to iterate without paying ~77s to re-read 418 MB.
func TestReaderFidelity(t *testing.T) {
if testing.Short() {
t.Skip("-short: skipping the 299-file corpus sweep")
t.Skip("-short: skipping the full-corpus sweep")
}
root := repoRoot(t)
manifest := filepath.Join(root, "parser", "testdata", "reader-fidelity-hashes.tsv")
+7 -4
View File
@@ -2,9 +2,12 @@
// contract, mirroring parser/src/reader.rs (which wraps calamine's
// open_workbook_auto for both formats).
//
// The contract is deliberately row-streaming rather than whole-workbook
// materialising: data/2017/ha-noi.xls alone holds 72k rows across two sheets,
// and the Rust original keeps at most one sheet range live at a time.
// The *contract* is row-at-a-time: callers receive one row per callback and
// never hold the workbook. The two implementations do not honour that in
// memory — both decode the whole workbook into [][][]Cell up front, because
// neither underlying library exposes a usable row cursor. data/2017/ha-noi.xls
// alone holds 72k rows across two sheets, so a caller that assumed streaming
// memory here would be wrong.
package reader
import (
@@ -27,7 +30,7 @@ type Cell struct {
// Sheet carries a sheet's identity and geometry. Height and Width describe the
// used range, matching calamine's Range::height()/width(); every used range in
// the corpus starts at (0,0), verified across all 299 files.
// the corpus starts at (0,0), verified across every input file.
type Sheet struct {
Index int
Name string
+1 -2
View File
@@ -1,8 +1,7 @@
// Package schema is the single source of truth for the SQL shape of every
// dataset — a direct port of parser/src/schema.rs.
//
// All four datasets (2016, 2017, 2017-old, 2017-old2) write into the same
// 22-column student table. Columns a dataset has no data for bind NULL.
// Every dataset writes into the same 22-column student table. Columns a dataset has no data for bind NULL.
//
// Column provenance:
//
+7 -25
View File
@@ -117,12 +117,12 @@ type Stats struct {
//
// VACUUM must run AFTER the transaction commits — SQLite refuses it inside one.
//
// The wording branches on datasetLabel, which Rust derives from the input
// directory's basename (main.rs:98-101). That makes the output depend on a
// filesystem path rather than on config; it is reproduced here for parity, and
// the caller passes the label explicitly so tests are not at the mercy of a
// temp-directory name.
func Finish(db *sql.DB, dbPath string, st Stats, datasetLabel string, isOld2 bool) error {
// The Rust original varied the wording by dataset, keying off the input
// directory's basename (main.rs:98-101) — one phrasing for the 2017-old2
// export, another for anything with "old" in its name. Both of those datasets
// have been removed, so only one branch was ever taken; the wording below is
// exactly what 2016 and 2017 have always printed.
func Finish(db *sql.DB, dbPath string, st Stats) error {
if _, err := db.Exec("VACUUM"); err != nil {
return fmt.Errorf("vacuum: %w", err)
}
@@ -135,22 +135,13 @@ func Finish(db *sql.DB, dbPath string, st Stats, datasetLabel string, isOld2 boo
insertable := st.SourceRows - st.Skipped
fmt.Println()
if isOld2 {
fmt.Printf("Source non-blank data rows: %d\n", st.SourceRows)
fmt.Printf(" skipped (empty/non-numeric SBD): %d\n", st.Skipped)
} else {
fmt.Printf("Source data rows (post-header): %d\n", st.SourceRows)
if containsOld(datasetLabel) {
fmt.Printf(" skipped (empty/non-numeric SBD): %d\n", st.Skipped)
} else {
fmt.Printf(" skipped (empty/invalid): %d\n", st.Skipped)
}
}
fmt.Printf(" insertable: %d\n", insertable)
fmt.Printf(" insert errors: %d\n", st.Errors)
fmt.Printf("DB rows (distinct SBD): %d\n", dbCount)
if !containsOld(datasetLabel) && st.Errors == 0 {
if st.Errors == 0 {
gap := int64(insertable) - dbCount
if gap == 0 {
fmt.Println("Audit: OK — every source row made it in.")
@@ -166,12 +157,3 @@ func Finish(db *sql.DB, dbPath string, st Stats, datasetLabel string, isOld2 boo
fmt.Printf("Size: %.1f MB\n", float64(size)/1024.0/1024.0)
return nil
}
func containsOld(label string) bool {
for i := 0; i+3 <= len(label); i++ {
if label[i:i+3] == "old" {
return true
}
}
return false
}
Binary file not shown.
+1 -1
View File
@@ -5,7 +5,7 @@
# FROZEN. These hashes were produced by the Rust/calamine parser, which has
# since been removed, so they cannot be regenerated. They remain a real
# regression guard: any change to the Go reader that alters a single cell of any
# of the 299 input files fails the suite. If the inputs themselves ever change,
# input file listed below fails the suite. If the inputs themselves ever change,
# the affected rows must be re-derived deliberately and reviewed, not refreshed
# in bulk.
data/2016/023718c7d3cf7ace3a7116fabb12bd9cdhyduoccantho-1468920829104.xlsx fc3afd8da9ed28b0fc177192fab7b19a06535cb45d86841835d8e8828fc61c1c
1 # Canonical cell-dump SHA-256 per input file, produced by the Rust/calamine
5 # FROZEN. These hashes were produced by the Rust/calamine parser, which has
6 # since been removed, so they cannot be regenerated. They remain a real
7 # regression guard: any change to the Go reader that alters a single cell of any
8 # of the 299 input files fails the suite. If the inputs themselves ever change, # input file listed below fails the suite. If the inputs themselves ever change,
9 # the affected rows must be re-derived deliberately and reviewed, not refreshed
10 # in bulk.
11 data/2016/023718c7d3cf7ace3a7116fabb12bd9cdhyduoccantho-1468920829104.xlsx
File diff suppressed because it is too large. Load diff
File diff suppressed because it is too large. Load diff
-85
View File
@@ -1,85 +0,0 @@
# Parser parity result
**Archived record.** This documents a verification run made on 2026-08-13,
against a tree that had a Rust parser, npm-script entry points and four
datasets — none of which exist now. It is kept because `docs/data-pipeline.md`
cites it as the evidence that the recovered foreign-language scores are real
rather than false regex matches. Do not expect the commands below to run.
Comparison of the databases the unified pipeline ships against a baseline built
from the pre-refactor code. Run on 2026-08-13 against the tree as it then stood
(`npm run build:rust && npm run build:db`), gate exit code 0.
Inputs:
- `plans/reports/parser-parity-baseline.json` — built with the two separate
crates, before any change
- `plans/reports/parser-parity-current.json` — decompressed from
`.build/public/db/*.db.gz`, i.e. the exact bytes published
## Result
| Dataset | Rows | Columns | Size |
| --- | --- | --- | --- |
| 2016 | 877,461 | 18 → 22 | +1.4% |
| 2017 | 861,068 | 18 → 22 | +2.2% |
| 2017-old | 847,348 | 18 → 22 | +2.2% |
| 2017-old2 | 679,764 | 18 → 22 | +2.2% |
Unchanged, for every dataset:
- row count
- non-NULL count for all 18 pre-existing columns
- every field of a deterministic student sample (SBDs ending `0000`)
## Approved recoveries
Unifying the subject regexes recovered 1,691 foreign-language scores that the
old per-year configs discarded. The 2016 config listed 12 subject patterns and
the 2017 configs 14; neither list was complete, so candidates who sat German,
Japanese or Russian ended up with no foreign-language score at all.
| Dataset | Recovered |
| --- | --- |
| 2016 | 182 × `tieng_nga` |
| 2017 | 93 × `tieng_duc`, 512 × `tieng_nhat` |
| 2017-old | 85 × `tieng_duc`, 484 × `tieng_nhat` |
| 2017-old2 | 22 × `tieng_duc`, 313 × `tieng_nhat` |
Confirmed real rather than spurious matches:
- Every student in all four datasets holds zero or exactly one foreign
language — never two — so nothing is duplicated.
- Each affected student had all language columns NULL beforehand. SBD
`01003198` went from all-NULL to `tieng_duc = 8`.
- Counts track dataset size across the three 2017 generations (93/85/22
German, 512/484/313 Japanese).
- Values are ordinary exam scores in the 0–10 range.
These exact counts are pinned as `APPROVED_RECOVERY` in
`parser/scripts/verify-parity.js`. Any other newly-populated column, or drift
in these numbers, still fails the gate.
## Columns that must stay empty
Verified 0 non-NULL:
- 2016: `khtn`, `khxh`, `gdcd`
- 2017, 2017-old, 2017-old2: `ten_cum_thi`, `gioi_tinh`
## Reproducing
```sh
npm run build:rust && npm run build:db
node parser/scripts/db-stats.js \
2016=<db> 2017=<db> 2017-old=<db> 2017-old2=<db> > current.json
node parser/scripts/verify-parity.js \
plans/reports/parser-parity-baseline.json current.json
```
The baseline cannot be regenerated — it required the two pre-refactor crates,
which no longer exist. It is committed for that reason.
## Unresolved questions
None.
+6 -2
View File
@@ -2,7 +2,6 @@
<html lang="vi">
<head>
<meta charset="UTF-8" />
<link rel="icon" type="image/svg+xml" href="/favicon.svg" />
<meta name="viewport" content="width=device-width, initial-scale=1.0" />
<link rel="preconnect" href="https://fonts.googleapis.com" />
<link rel="preconnect" href="https://fonts.gstatic.com" crossorigin />
@@ -10,7 +9,12 @@
rel="stylesheet"
href="https://fonts.googleapis.com/css2?family=Be+Vietnam+Pro:wght@400;500;600;700&display=swap"
/>
<title>Tra cứu điểm thi THPT QG 2017</title>
<!--
One index.html is copied to every route, so this title has to be the one
that suits them all; the app replaces it with the dataset's own title once
it knows which route it is on.
-->
<title>Tra cứu điểm thi THPT Quốc gia</title>
</head>
<body>
<div id="root"></div>
-11
View File
@@ -14,7 +14,6 @@
--focus-ring: 0 0 0 3px rgba(59, 79, 228, 0.35);
--radius: 10px;
--radius-sm: 6px;
--shadow-sm: 0 1px 2px rgba(15, 20, 40, 0.05);
--shadow-md: 0 6px 20px rgba(15, 20, 40, 0.08);
/* Score tier palette — TFT rarity ladder (white → green → blue → purple → gold → prismatic).
@@ -49,7 +48,6 @@
--primary-hover: #879aff;
--primary-ink: #0f1220;
--focus-ring: 0 0 0 3px rgba(109, 130, 255, 0.45);
--shadow-sm: 0 1px 2px rgba(0, 0, 0, 0.4);
--shadow-md: 0 6px 20px rgba(0, 0, 0, 0.55);
--tier-common-bg: #242838; --tier-common-ink: #c5cad8; --tier-common-accent: #4a5163;
@@ -853,16 +851,7 @@ footer {
font-size: 0.9rem;
}
.hub-archive h2 {
font-size: 0.95rem;
font-weight: 600;
color: var(--text-muted);
margin-bottom: 0;
}
.hub-list-compact {
margin-top: 0.75rem;
}
.hub-back {
margin: 0.35rem 0 0;
+6
View File
@@ -147,6 +147,12 @@ function DatasetApp({ dataset }) {
writeUrlQuery("");
}, []);
// The assembler copies one index.html to every route, so its static <title>
// cannot name a dataset. Set the real one once the route is resolved.
useEffect(() => {
document.title = dataset.title;
}, [dataset.title]);
return (
<div className="app">
<header>
-1
View File
@@ -1 +0,0 @@
<svg xmlns="http://www.w3.org/2000/svg" width="77" height="47" fill="none" aria-labelledby="vite-logo-title" viewBox="0 0 77 47"><title id="vite-logo-title">Vite</title><style>.parenthesis{fill:#000}@media (prefers-color-scheme:dark){.parenthesis{fill:#fff}}</style><path fill="#9135ff" d="M40.151 45.71c-.663.844-2.02.374-2.02-.699V34.708a2.26 2.26 0 0 0-2.262-2.262H24.493c-.92 0-1.457-1.04-.92-1.788l7.479-10.471c1.07-1.498 0-3.578-1.842-3.578H15.443c-.92 0-1.456-1.04-.92-1.788l9.696-13.576c.213-.297.556-.474.92-.474h28.894c.92 0 1.456 1.04.92 1.788l-7.48 10.472c-1.07 1.497 0 3.578 1.842 3.578h11.376c.944 0 1.474 1.087.89 1.83L40.153 45.712z"/><mask id="a" width="48" height="47" x="14" y="0" maskUnits="userSpaceOnUse" style="mask-type:alpha"><path fill="#000" d="M40.047 45.71c-.663.843-2.02.374-2.02-.699V34.708a2.26 2.26 0 0 0-2.262-2.262H24.389c-.92 0-1.457-1.04-.92-1.788l7.479-10.472c1.07-1.497 0-3.578-1.842-3.578H15.34c-.92 0-1.456-1.04-.92-1.788l9.696-13.575c.213-.297.556-.474.92-.474H53.93c.92 0 1.456 1.04.92 1.788L47.37 13.03c-1.07 1.498 0 3.578 1.842 3.578h11.376c.944 0 1.474 1.088.89 1.831L40.049 45.712z"/></mask><g mask="url(#a)"><g filter="url(#b)"><ellipse cx="5.508" cy="14.704" fill="#eee6ff" rx="5.508" ry="14.704" transform="rotate(269.814 20.96 11.29)scale(-1 1)"/></g><g filter="url(#c)"><ellipse cx="10.399" cy="29.851" fill="#eee6ff" rx="10.399" ry="29.851" transform="rotate(89.814 -16.902 -8.275)scale(1 -1)"/></g><g filter="url(#d)"><ellipse cx="5.508" cy="30.487" fill="#8900ff" rx="5.508" ry="30.487" transform="rotate(89.814 -19.197 -7.127)scale(1 -1)"/></g><g filter="url(#e)"><ellipse cx="5.508" cy="30.599" fill="#8900ff" rx="5.508" ry="30.599" transform="rotate(89.814 -25.928 4.177)scale(1 -1)"/></g><g filter="url(#f)"><ellipse cx="5.508" cy="30.599" fill="#8900ff" rx="5.508" ry="30.599" transform="rotate(89.814 -25.738 5.52)scale(1 -1)"/></g><g filter="url(#g)"><ellipse cx="14.072" cy="22.078" fill="#eee6ff" rx="14.072" ry="22.078" transform="rotate(93.35 31.245 55.578)scale(-1 1)"/></g><g filter="url(#h)"><ellipse cx="3.47" cy="21.501" fill="#8900ff" rx="3.47" ry="21.501" transform="rotate(89.009 35.419 55.202)scale(-1 1)"/></g><g filter="url(#i)"><ellipse cx="3.47" cy="21.501" fill="#8900ff" rx="3.47" ry="21.501" transform="rotate(89.009 35.419 55.202)scale(-1 1)"/></g><g filter="url(#j)"><ellipse cx="14.592" cy="9.743" fill="#8900ff" rx="4.407" ry="29.108" transform="rotate(39.51 14.592 9.743)"/></g><g filter="url(#k)"><ellipse cx="61.728" cy="-5.321" fill="#8900ff" rx="4.407" ry="29.108" transform="rotate(37.892 61.728 -5.32)"/></g><g filter="url(#l)"><ellipse cx="55.618" cy="7.104" fill="#00c2ff" rx="5.971" ry="9.665" transform="rotate(37.892 55.618 7.104)"/></g><g filter="url(#m)"><ellipse cx="12.326" cy="39.103" fill="#8900ff" rx="4.407" ry="29.108" transform="rotate(37.892 12.326 39.103)"/></g><g filter="url(#n)"><ellipse cx="12.326" cy="39.103" fill="#8900ff" rx="4.407" ry="29.108" transform="rotate(37.892 12.326 39.103)"/></g><g filter="url(#o)"><ellipse cx="49.857" cy="30.678" fill="#8900ff" rx="4.407" ry="29.108" transform="rotate(37.892 49.857 30.678)"/></g><g filter="url(#p)"><ellipse cx="52.623" cy="33.171" fill="#00c2ff" rx="5.971" ry="15.297" transform="rotate(37.892 52.623 33.17)"/></g></g><path d="M6.919 0c-9.198 13.166-9.252 33.575 0 46.789h6.215c-9.25-13.214-9.196-33.623 0-46.789zm62.424 0h-6.215c9.198 13.166 9.252 33.575 0 46.789h6.215c9.25-13.214 9.196-33.623 0-46.789" class="parenthesis"/><defs><filter id="b" width="60.045" height="41.654" x="-5.564" y="16.92" color-interpolation-filters="sRGB" filterUnits="userSpaceOnUse"><feFlood flood-opacity="0" result="BackgroundImageFix"/><feBlend in="SourceGraphic" in2="BackgroundImageFix" result="shape"/><feGaussianBlur result="effect1_foregroundBlur_2002_17286" stdDeviation="7.659"/></filter><filter id="c" width="90.34" height="51.437" x="-40.407" y="-6.762" color-interpolation-filters="sRGB" filterUnits="userSpaceOnUse"><feFlood flood-opacity="0" result="BackgroundImageFix"/><feBlend in="SourceGraphic" in2="BackgroundImageFix" result="shape"/><feGaussianBlur result="effect1_foregroundBlur_2002_17286" stdDeviation="7.659"/></filter><filter id="d" width="79.355" height="29.4" x="-35.435" y="2.801" color-interpolation-filters="sRGB" filterUnits="userSpaceOnUse"><feFlood flood-opacity="0" result="BackgroundImageFix"/><feBlend in="SourceGraphic" in2="BackgroundImageFix" result="shape"/><feGaussianBlur result="effect1_foregroundBlur_2002_17286" stdDeviation="4.596"/></filter><filter id="e" width="79.579" height="29.4" x="-30.84" y="20.8" color-interpolation-filters="sRGB" filterUnits="userSpaceOnUse"><feFlood flood-opacity="0" result="BackgroundImageFix"/><feBlend in="SourceGraphic" in2="BackgroundImageFix" result="shape"/><feGaussianBlur result="effect1_foregroundBlur_2002_17286" stdDeviation="4.596"/></filter><filter id="f" width="79.579" height="29.4" x="-29.307" y="21.949" color-interpolation-filters="sRGB" filterUnits="usLine truncated

Before

Width:  |  Height:  |  Size: 8.5 KiB

+1 -17
View File
@@ -8,8 +8,6 @@ import { DATASETS, pathOf } from "../datasets";
* No database is fetched on this route.
*/
export function Hub() {
const current = DATASETS.filter((d) => !d.id.includes("old"));
const archived = DATASETS.filter((d) => d.id.includes("old"));
const base = import.meta.env.BASE_URL;
return (
@@ -24,7 +22,7 @@ export function Hub() {
<main>
<ul className="hub-list" role="list">
{current.map((d) => (
{DATASETS.map((d) => (
<li key={d.id}>
<a className="hub-link" href={pathOf(d, base)}>
<span className="hub-label">{d.label}</span>
@@ -33,20 +31,6 @@ export function Hub() {
</li>
))}
</ul>
<section className="hub-archive" aria-labelledby="archive-heading">
<h2 id="archive-heading">Phiên bản cũ của trang 2017</h2>
<ul className="hub-list hub-list-compact" role="list">
{archived.map((d) => (
<li key={d.id}>
<a className="hub-link" href={pathOf(d, base)}>
<span className="hub-label">{d.label}</span>
<span className="hub-blurb">{d.blurb}</span>
</a>
</li>
))}
</ul>
</section>
</main>
<footer>
+3 -5
View File
@@ -24,7 +24,6 @@ const SUBTITLE = "Dữ liệu thí sinh toàn quốc · Hỗ trợ truy vấn SQ
const CONTENT = {
2016: {
label: "Kỳ thi 2016",
blurb: "877.461 thí sinh",
title: "Tra cứu điểm thi THPT Quốc gia 2016",
subtitle: SUBTITLE,
source: "Bộ GD&ĐT",
@@ -33,7 +32,6 @@ const CONTENT = {
},
2017: {
label: "Kỳ thi 2017",
blurb: "861.068 thí sinh",
title: "Tra cứu điểm thi THPT Quốc gia 2017",
subtitle: SUBTITLE,
source: "baotintuc.vn",
@@ -60,12 +58,12 @@ for (const id of Object.keys(CONTENT)) {
export const DATASETS = registry.datasets.map((d) => ({
id: d.id,
dbSizeMb: d.dbSizeMb,
// Derived, not written twice: the candidate count the hub shows is the same
// number the assembler enforces, so the two cannot drift apart.
blurb: `${d.expectedRows.toLocaleString("vi-VN")} thí sinh`,
...CONTENT[d.id],
}));
/** Dataset IDs in build order. */
export const DATASET_IDS = DATASETS.map((d) => d.id);
/** Site path for a dataset, e.g. pathOf(d, "/thptqg/") → "/thptqg/2017/". */
export function pathOf(dataset, base) {
return `${base}${dataset.id}/`;
+19 -23
View File
@@ -1,7 +1,7 @@
/**
* The 16 subject columns of the canonical schema, in display order.
*
* Mirrors `SCORE_FIELDS` in parser/internal/schema/schema.go. Previously this list was
* Mirrors `ScoreFields` in parser/internal/schema/schema.go. Previously this list was
* maintained separately in score-table.jsx and student-detail.jsx, which is how
* they drifted out of sync with each other and with the database.
*
@@ -11,29 +11,29 @@
*/
export const SUBJECTS = [
{ key: "toan", label: "Toán", short: "Toán" },
{ key: "ngu_van", label: "Ngữ văn", short: "Văn" },
{ key: "vat_ly", label: "Vật lí", short: "Lý" },
{ key: "hoa_hoc", label: "Hóa học", short: "Hóa" },
{ key: "sinh_hoc", label: "Sinh học", short: "Sinh" },
{ key: "khtn", label: "KHTN", short: "KHTN" },
{ key: "lich_su", label: "Lịch sử", short: "Sử" },
{ key: "dia_ly", label: "Địa lí", short: "Địa" },
{ key: "gdcd", label: "GDCD", short: "GDCD" },
{ key: "khxh", label: "KHXH", short: "KHXH" },
{ key: "tieng_anh", label: "Tiếng Anh", short: "T.Anh" },
{ key: "tieng_phap", label: "Tiếng Pháp", short: "T.Pháp" },
{ key: "tieng_nga", label: "Tiếng Nga", short: "T.Nga" },
{ key: "tieng_duc", label: "Tiếng Đức", short: "T.Đức" },
{ key: "tieng_nhat", label: "Tiếng Nhật", short: "T.Nhật" },
{ key: "tieng_trung", label: "Tiếng Trung", short: "T.Trung" },
{ key: "toan", label: "Toán" },
{ key: "ngu_van", label: "Ngữ văn" },
{ key: "vat_ly", label: "Vật lí" },
{ key: "hoa_hoc", label: "Hóa học" },
{ key: "sinh_hoc", label: "Sinh học" },
{ key: "khtn", label: "KHTN" },
{ key: "lich_su", label: "Lịch sử" },
{ key: "dia_ly", label: "Địa lí" },
{ key: "gdcd", label: "GDCD" },
{ key: "khxh", label: "KHXH" },
{ key: "tieng_anh", label: "Tiếng Anh" },
{ key: "tieng_phap", label: "Tiếng Pháp" },
{ key: "tieng_nga", label: "Tiếng Nga" },
{ key: "tieng_duc", label: "Tiếng Đức" },
{ key: "tieng_nhat", label: "Tiếng Nhật" },
{ key: "tieng_trung", label: "Tiếng Trung" },
];
/**
* Non-score columns worth showing in the results table.
*
* Only the 2016 dataset populates these; they are NULL throughout the 2017
* datasets and get hidden by the same all-NULL filter that hides unused
* Only the 2016 dataset populates these; they are NULL throughout 2017
* and get hidden by the same all-NULL filter that hides unused
* subjects, so no per-dataset conditional is needed.
*/
export const IDENTITY_COLUMNS = [
@@ -41,10 +41,6 @@ export const IDENTITY_COLUMNS = [
{ key: "gioi_tinh", label: "GT" },
];
export const SUBJECT_LABELS = Object.fromEntries(
SUBJECTS.map((s) => [s.key, s.label]),
);
/** True when at least one row carries a value for `key`. */
export function hasAnyValue(rows, key) {
return rows.some((row) => row[key] !== null && row[key] !== undefined);
+4 -4
View File
@@ -10,8 +10,8 @@ import react from "@vitejs/plugin-react";
// path, so every URL is a real static file and no SPA 404-fallback is needed —
// and the existing ?q= deep links keep working, which that fallback would break.
//
// publicDir holds only the gzipped databases, staged there by
// parser/scripts/build-db.js. Nothing uncompressed is ever placed in it.
// publicDir holds only the gzipped databases, staged there by the assembler
// (assembler/internal/databases). Nothing uncompressed is ever placed in it.
//
// It sits at the repository root rather than inside this workspace because
// parser writes it, so the path has to climb out of web/. Vite resolves
@@ -19,8 +19,8 @@ import react from "@vitejs/plugin-react";
//
// outDir is left at its default, so the build lands in web/dist and this
// workspace stays self-contained. The Pages artifact (_site) is assembled at
// the repository root by web/scripts/assemble-site.js, since that is where the
// deploy action uploads from.
// the repository root by the assembler (assembler/internal/site), since that is
// where the deploy action uploads from.
export default defineConfig({
plugins: [react()],
base: "/thptqg/",