From a0420fdc371541cecc9c1404af61b6d167c56b8e Mon Sep 17 00:00:00 2001 From: tiennm99 Date: Thu, 13 Aug 2026 23:48:25 +0700 Subject: [PATCH] refactor: remove the last JS script and the dead weight three audits found MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The pipeline is now Go outside web/. differential-parity.mjs becomes assembler/internal/verify, reachable as `assemble verify A B`. The port fixed a real weakness: the JavaScript hashed each row's fields joined bare, so a value shifted across a column boundary produced the same digest. A test now pins that. The hub still rendered "Phiên bản cũ của trang 2017" above a permanently empty list — it split datasets on id.includes("old"), and both such datasets are gone. The heading and the filter are removed. index.html titled every page "THPT QG 2017", including 2016 and the hub, because one file is copied to every route; the static title is now neutral and the app sets the dataset's own. Dead code removed: the isOld2/containsOld branches in the stats block, which only 2017-old2 could ever reach; SUBJECT_LABELS, DATASET_IDS and the unread `short` subject field; an unused vite.svg and a favicon link to a file that never existed; two unused CSS rules and --shadow-sm; site.Paths.Root. Corrected comments that were confidently wrong rather than merely stale: the reader claimed to be row-streaming when both implementations decode the whole workbook into memory first, and the fidelity oracle still spoke of 299 input files when it covers 182. Candidate counts in the hub now derive from datasets.json instead of being written a second time as prose. plans/ is emptied. The parity report it held was cited by docs/data-pipeline.md, so the evidence that the recovered foreign-language scores are real — not the citation, the four arguments themselves — is now inline there. Verified: 2017 rebuilt after the writer change hashes identically to the build before it. --- .github/workflows/deploy-pages.yml | 6 + README.md | 1 + assembler/cmd/assemble/main.go | 76 +- assembler/internal/site/site.go | 2 - assembler/internal/site/site_test.go | 4 +- assembler/internal/verify/verify.go | 384 ++ assembler/internal/verify/verify_test.go | 165 + crawler/internal/sources/sources_test.go | 2 +- docs/data-pipeline.md | 57 +- docs/deployment-guide.md | 4 +- parser/README.md | 11 +- parser/cmd/dumpcells/main.go | 11 +- parser/cmd/xlsxread/main.go | 4 +- parser/internal/config/config.go | 4 +- parser/internal/ingest/detect2016.go | 2 +- parser/internal/ingest/ingest.go | 3 +- parser/internal/reader/fidelity_test.go | 2 +- parser/internal/reader/reader.go | 11 +- parser/internal/schema/schema.go | 3 +- parser/internal/writer/writer.go | 36 +- parser/scripts/differential-parity.mjs | Bin 7594 -> 0 bytes parser/testdata/reader-fidelity-hashes.tsv | 2 +- plans/reports/parser-parity-baseline.json | 3910 ---------------- plans/reports/parser-parity-current.json | 4686 -------------------- plans/reports/parser-parity-result.md | 85 - web/index.html | 8 +- web/src/App.css | 11 - web/src/App.jsx | 6 + web/src/assets/vite.svg | 1 - web/src/components/hub.jsx | 18 +- web/src/datasets.js | 8 +- web/src/lib/subjects.js | 42 +- web/vite.config.js | 8 +- 33 files changed, 754 insertions(+), 8819 deletions(-) create mode 100644 assembler/internal/verify/verify.go create mode 100644 assembler/internal/verify/verify_test.go delete mode 100644 parser/scripts/differential-parity.mjs delete mode 100644 plans/reports/parser-parity-baseline.json delete mode 100644 plans/reports/parser-parity-current.json delete mode 100644 plans/reports/parser-parity-result.md delete mode 100644 web/src/assets/vite.svg diff --git a/.github/workflows/deploy-pages.yml b/.github/workflows/deploy-pages.yml index a8ae901..ef85e05 100644 --- a/.github/workflows/deploy-pages.yml +++ b/.github/workflows/deploy-pages.yml @@ -62,6 +62,12 @@ jobs: - name: Test crawler run: go -C crawler test ./... + # The assembler owns every guard between a built database and the + # published site, so its tests are the ones that prove a short or missing + # database cannot ship. + - name: Test assembler + run: go -C assembler test ./... + - name: Lint web working-directory: web run: npm run lint diff --git a/README.md b/README.md index 627f52a..a500512 100644 --- a/README.md +++ b/README.md @@ -59,6 +59,7 @@ is missing altogether. Sub-steps when iterating: ```bash go -C assembler run ./cmd/assemble db 2017 # one database go -C assembler run ./cmd/assemble site # web build and _site only +go -C assembler run ./cmd/assemble verify A B # compare two sets of databases (cd web && npm run dev) # the app against staged databases ``` diff --git a/assembler/cmd/assemble/main.go b/assembler/cmd/assemble/main.go index adaa95b..6d2a898 100644 --- a/assembler/cmd/assemble/main.go +++ b/assembler/cmd/assemble/main.go @@ -4,6 +4,7 @@ // assemble # databases, then the site // assemble db # databases only (add ids to limit: assemble db 2017) // assemble site # web build and _site only, reusing staged databases +// assemble verify A B # compare two sets of built databases field by field // // It sequences the other stages rather than doing their work: the parser reads // spreadsheets, Vite bundles the app, and this decides what runs, checks what @@ -18,6 +19,7 @@ import ( "github.com/tiennm99/thptqg/assembler/internal/databases" "github.com/tiennm99/thptqg/assembler/internal/registry" "github.com/tiennm99/thptqg/assembler/internal/site" + "github.com/tiennm99/thptqg/assembler/internal/verify" ) func main() { @@ -31,12 +33,23 @@ func usage() { fmt.Fprint(os.Stderr, `usage: assemble [step] [dataset...] Steps: - (none) databases, then the site - db build, verify and compress the databases - site build the web app and assemble _site + (none) databases, then the site + db build, verify and compress the databases + site build the web app and assemble _site + verify A B compare two directories of built databases, field by field Naming datasets limits the db step to those; the site step always covers all of them, since a partial site would publish links to databases it did not build. + +verify takes two directories holding .db.gz (or .db) — for example a +copy of the previous .build/public/db against a fresh build. Relative paths +resolve against the repository root, not the working directory. It reports row +counts, per-column non-NULL counts, a full-table hash and the first differing +rows. + +Nothing else checks database content: the reader oracle covers reading and the +row-count guard covers how many rows, but a change in transform or writer logic +can alter what is in them while both still pass. `) } @@ -44,7 +57,7 @@ func run(args []string) error { step := "" if len(args) > 0 { switch args[0] { - case "db", "site": + case "db", "site", "verify": step, args = args[0], args[1:] case "-h", "--help": usage() @@ -66,6 +79,19 @@ func run(args []string) error { return err } + if step == "verify" { + if len(args) != 2 { + usage() + return fmt.Errorf("verify needs two directories") + } + // Relative paths resolve against the repository root, not the working + // directory. The documented invocation is `go -C assembler run ...`, + // and -C moves the process into assembler/ before main starts — so a + // bare `.build/public/db` would otherwise look for it in there and + // report a missing database that is sitting in plain sight. + return runVerify(all, atRoot(root, args[0]), atRoot(root, args[1])) + } + if step == "" || step == "db" { selected, err := registry.Select(all, args) if err != nil { @@ -88,6 +114,48 @@ func run(args []string) error { return nil } +// atRoot resolves a relative path against the repository root. +func atRoot(root, path string) string { + if filepath.IsAbs(path) { + return path + } + return filepath.Join(root, path) +} + +// runVerify compares two sets of built databases and fails loudly on any +// difference. A dataset missing from either side is an error rather than a +// skip: silently comparing one of two datasets is how a gate passes without +// proving anything. +func runVerify(all []registry.Dataset, dirA, dirB string) error { + results, err := verify.Compare(all, dirA, dirB) + if err != nil { + return err + } + + bad := 0 + for _, r := range results { + fmt.Printf("--- %s ---\n", r.ID) + fmt.Printf(" rows : %d vs %d\n", r.RowsA, r.RowsB) + if r.OK() { + fmt.Printf(" full-table : identical sha256 %s\n", r.HashA[:16]) + continue + } + bad++ + for _, p := range r.Problems { + fmt.Printf(" *** %s\n", p) + } + for _, d := range r.FirstDiff { + fmt.Println(d) + } + } + + if bad > 0 { + return fmt.Errorf("%d dataset(s) differ", bad) + } + fmt.Println("\nIdentical — rows, per-column non-NULL counts, schema and full-table hash.") + return nil +} + func buildDatabases(root string, all, selected []registry.Dataset) error { p := databases.DefaultPaths(root) diff --git a/assembler/internal/site/site.go b/assembler/internal/site/site.go index f799bb4..75d2344 100644 --- a/assembler/internal/site/site.go +++ b/assembler/internal/site/site.go @@ -26,7 +26,6 @@ import ( // Paths locates the pieces this package needs. type Paths struct { - Root string // Web is the Vite project directory. Web string // Dist is where Vite emits, inside the web workspace. @@ -39,7 +38,6 @@ type Paths struct { func DefaultPaths(root string) Paths { web := filepath.Join(root, "web") return Paths{ - Root: root, Web: web, Dist: filepath.Join(web, "dist"), Site: filepath.Join(root, "_site"), diff --git a/assembler/internal/site/site_test.go b/assembler/internal/site/site_test.go index 33b0d67..0460d96 100644 --- a/assembler/internal/site/site_test.go +++ b/assembler/internal/site/site_test.go @@ -22,7 +22,7 @@ func fakeBuild(t *testing.T, dbs ...string) Paths { for _, name := range dbs { write(t, filepath.Join(dist, "db", name), "gzipped-bytes") } - return Paths{Root: root, Web: filepath.Join(root, "web"), Dist: dist, Site: filepath.Join(root, "_site")} + return Paths{Web: filepath.Join(root, "web"), Dist: dist, Site: filepath.Join(root, "_site")} } func write(t *testing.T, path, body string) { @@ -113,7 +113,7 @@ func TestGzipIsNotMistakenForRaw(t *testing.T) { func TestAssembleRejectsAMissingBuild(t *testing.T) { root := t.TempDir() - p := Paths{Root: root, Web: root, Dist: filepath.Join(root, "dist"), Site: filepath.Join(root, "_site")} + p := Paths{Web: root, Dist: filepath.Join(root, "dist"), Site: filepath.Join(root, "_site")} if err := Assemble(p, datasets); err == nil { t.Fatal("expected an error when there is no Vite build") } diff --git a/assembler/internal/verify/verify.go b/assembler/internal/verify/verify.go new file mode 100644 index 0000000..76c0892 --- /dev/null +++ b/assembler/internal/verify/verify.go @@ -0,0 +1,384 @@ +// Package verify compares two sets of built databases field by field. +// +// Nothing else checks database *content*. The reader-fidelity oracle covers +// reading the spreadsheets, and the row-count guard covers how many rows came +// out — but a change in transform or writer logic can alter what is in those +// rows while both of those still pass. This is what catches that. +// +// It was written as the gate for the Rust-to-Go parser migration and works for +// any two builds: before and after a refactor, or a re-crawl against the +// databases already published. +package verify + +import ( + "compress/gzip" + "crypto/sha256" + "database/sql" + "encoding/hex" + "fmt" + "io" + "math" + "os" + "path/filepath" + "sort" + "strconv" + "strings" + + _ "modernc.org/sqlite" + + "github.com/tiennm99/thptqg/assembler/internal/registry" +) + +const driverName = "sqlite" + +// Result is the outcome for one dataset. +type Result struct { + ID string + Problems []string + RowsA int64 + RowsB int64 + HashA string + HashB string + FirstDiff []string +} + +// OK reports whether the two databases are logically identical. +func (r Result) OK() bool { return len(r.Problems) == 0 } + +// Compare checks every dataset in the registry, reading /.db.gz (or +// .db) from each side. +func Compare(datasets []registry.Dataset, dirA, dirB string) ([]Result, error) { + out := make([]Result, 0, len(datasets)) + for _, d := range datasets { + r, err := compareOne(d.ID, dirA, dirB) + if err != nil { + return nil, err + } + out = append(out, r) + } + return out, nil +} + +func compareOne(id, dirA, dirB string) (Result, error) { + res := Result{ID: id} + + a, cleanA, err := open(dirA, id) + if err != nil { + return res, err + } + defer cleanA() + defer a.Close() + + b, cleanB, err := open(dirB, id) + if err != nil { + return res, err + } + defer cleanB() + defer b.Close() + + sigA, err := schemaSignature(a) + if err != nil { + return res, err + } + sigB, err := schemaSignature(b) + if err != nil { + return res, err + } + if sigA != sigB { + res.Problems = append(res.Problems, "schema/index metadata differs") + } + + cols, err := columns(a) + if err != nil { + return res, err + } + + if res.RowsA, err = count(a, "SELECT COUNT(*) FROM student"); err != nil { + return res, err + } + if res.RowsB, err = count(b, "SELECT COUNT(*) FROM student"); err != nil { + return res, err + } + if res.RowsA != res.RowsB { + res.Problems = append(res.Problems, fmt.Sprintf("row count %d vs %d", res.RowsA, res.RowsB)) + } + + // Per-column non-NULL counts localise a difference to a column even when the + // full-table hash has already said "something differs". + for _, c := range cols { + q := fmt.Sprintf("SELECT COUNT(%s) FROM student", c) + na, err := count(a, q) + if err != nil { + return res, err + } + nb, err := count(b, q) + if err != nil { + return res, err + } + if na != nb { + res.Problems = append(res.Problems, fmt.Sprintf("non-NULL count for %s: %d vs %d", c, na, nb)) + } + } + + if res.HashA, err = tableHash(a, cols); err != nil { + return res, err + } + if res.HashB, err = tableHash(b, cols); err != nil { + return res, err + } + if res.HashA != res.HashB { + res.Problems = append(res.Problems, "full-table hash differs") + if res.FirstDiff, err = firstDifferences(a, b, cols, 20); err != nil { + return res, err + } + } + + return res, nil +} + +// open finds .db.gz or .db in dir and returns a handle. A compressed +// database is expanded to a temporary file, since SQLite needs to seek. +func open(dir, id string) (*sql.DB, func(), error) { + noop := func() {} + + plain := filepath.Join(dir, id+".db") + if _, err := os.Stat(plain); err == nil { + db, err := sql.Open(driverName, "file:"+plain+"?mode=ro") + return db, noop, err + } + + gzPath := filepath.Join(dir, id+".db.gz") + f, err := os.Open(gzPath) + if err != nil { + return nil, noop, fmt.Errorf("no %s.db or %s.db.gz in %s", id, id, dir) + } + defer f.Close() + + zr, err := gzip.NewReader(f) + if err != nil { + return nil, noop, fmt.Errorf("%s: %w", gzPath, err) + } + defer zr.Close() + + tmp, err := os.CreateTemp("", "verify-"+id+"-*.db") + if err != nil { + return nil, noop, err + } + cleanup := func() { os.Remove(tmp.Name()) } + + if _, err := io.Copy(tmp, zr); err != nil { + tmp.Close() + cleanup() + return nil, noop, err + } + if err := tmp.Close(); err != nil { + cleanup() + return nil, noop, err + } + + db, err := sql.Open(driverName, "file:"+tmp.Name()+"?mode=ro") + if err != nil { + cleanup() + return nil, noop, err + } + return db, cleanup, nil +} + +func count(db *sql.DB, query string) (int64, error) { + var n int64 + err := db.QueryRow(query).Scan(&n) + return n, err +} + +// columns returns the column names in declaration order. +func columns(db *sql.DB) ([]string, error) { + rows, err := db.Query("SELECT name FROM pragma_table_info('student') ORDER BY cid") + if err != nil { + return nil, err + } + defer rows.Close() + + var out []string + for rows.Next() { + var name string + if err := rows.Scan(&name); err != nil { + return nil, err + } + out = append(out, name) + } + if err := rows.Err(); err != nil { + return nil, err + } + if len(out) == 0 { + return nil, fmt.Errorf("no student table") + } + return out, nil +} + +// schemaSignature reduces the table and index metadata to a comparable string. +func schemaSignature(db *sql.DB) (string, error) { + rows, err := db.Query("SELECT cid, name, type, \"notnull\", pk FROM pragma_table_info('student') ORDER BY cid") + if err != nil { + return "", err + } + var cols []string + for rows.Next() { + var cid, notnull, pk int + var name, typ string + if err := rows.Scan(&cid, &name, &typ, ¬null, &pk); err != nil { + rows.Close() + return "", err + } + cols = append(cols, fmt.Sprintf("%d:%s:%s:%d:%d", cid, name, typ, notnull, pk)) + } + rows.Close() + if err := rows.Err(); err != nil { + return "", err + } + + idxRows, err := db.Query("SELECT name, \"unique\", partial FROM pragma_index_list('student')") + if err != nil { + return "", err + } + defer idxRows.Close() + var idx []string + for idxRows.Next() { + var name string + var uniq, partial int + if err := idxRows.Scan(&name, &uniq, &partial); err != nil { + return "", err + } + idx = append(idx, fmt.Sprintf("%s:%d:%d", name, uniq, partial)) + } + if err := idxRows.Err(); err != nil { + return "", err + } + // Index order is not guaranteed by SQLite; sort so it cannot cause a false + // difference. + sort.Strings(idx) + + return strings.Join(cols, "|") + "\n" + strings.Join(idx, "|"), nil +} + +// Field and record separators. The values are hashed with explicit delimiters +// so that ("ab","c") and ("a","bc") cannot produce the same digest — joining +// them bare would make a shifted field boundary invisible. +const ( + fieldSep = "\x1f" + recordSep = "\x1e" +) + +// ser renders one value canonically. +// +// NULL gets a sentinel no real value can collide with, and a whole-numbered +// REAL is rendered with one decimal place so that a column holding 3 and one +// holding 3.0 compare equal, as SQLite considers them. +func ser(v any) string { + switch t := v.(type) { + case nil: + return "\x00NULL" + case int64: + return strconv.FormatInt(t, 10) + case float64: + if t == math.Trunc(t) && !math.IsInf(t, 0) { + return strconv.FormatFloat(t, 'f', 1, 64) + } + return strconv.FormatFloat(t, 'g', -1, 64) + case []byte: + return string(t) + case string: + return t + default: + return fmt.Sprint(t) + } +} + +// tableHash is a rolling SHA-256 over every row, ordered by primary key. +// +// Streamed rather than materialised: 877k rows by 22 columns would otherwise be +// a large amount of memory for no benefit. +func tableHash(db *sql.DB, cols []string) (string, error) { + rows, err := db.Query("SELECT * FROM student ORDER BY so_bao_danh") + if err != nil { + return "", err + } + defer rows.Close() + + h := sha256.New() + vals := make([]any, len(cols)) + ptrs := make([]any, len(cols)) + for i := range vals { + ptrs[i] = &vals[i] + } + + for rows.Next() { + if err := rows.Scan(ptrs...); err != nil { + return "", err + } + for i := range vals { + io.WriteString(h, ser(vals[i])) + io.WriteString(h, fieldSep) + } + io.WriteString(h, recordSep) + } + if err := rows.Err(); err != nil { + return "", err + } + return hex.EncodeToString(h.Sum(nil)), nil +} + +// firstDifferences walks both tables in step and reports where they diverge, +// so a failure names a row and a column rather than only a digest. +func firstDifferences(a, b *sql.DB, cols []string, limit int) ([]string, error) { + ra, err := a.Query("SELECT * FROM student ORDER BY so_bao_danh") + if err != nil { + return nil, err + } + defer ra.Close() + rb, err := b.Query("SELECT * FROM student ORDER BY so_bao_danh") + if err != nil { + return nil, err + } + defer rb.Close() + + scan := func(rows *sql.Rows) ([]string, bool, error) { + if !rows.Next() { + return nil, false, rows.Err() + } + vals := make([]any, len(cols)) + ptrs := make([]any, len(cols)) + for i := range vals { + ptrs[i] = &vals[i] + } + if err := rows.Scan(ptrs...); err != nil { + return nil, false, err + } + out := make([]string, len(cols)) + for i := range vals { + out[i] = ser(vals[i]) + } + return out, true, nil + } + + var out []string + for len(out) < limit { + va, okA, err := scan(ra) + if err != nil { + return nil, err + } + vb, okB, err := scan(rb) + if err != nil { + return nil, err + } + if !okA || !okB { + break + } + for i, c := range cols { + if va[i] != vb[i] { + out = append(out, fmt.Sprintf(" so_bao_danh=%s %s: a=%q b=%q", va[0], c, va[i], vb[i])) + break + } + } + } + return out, nil +} diff --git a/assembler/internal/verify/verify_test.go b/assembler/internal/verify/verify_test.go new file mode 100644 index 0000000..c367e51 --- /dev/null +++ b/assembler/internal/verify/verify_test.go @@ -0,0 +1,165 @@ +package verify + +import ( + "database/sql" + "os" + "path/filepath" + "strings" + "testing" + + "github.com/tiennm99/thptqg/assembler/internal/registry" +) + +// makeDB writes a miniature student table so the comparison logic can be +// exercised without building a real 877k-row database. +func makeDB(t *testing.T, dir, id string, rows [][3]any) { + t.Helper() + path := filepath.Join(dir, id+".db") + if err := os.MkdirAll(dir, 0o755); err != nil { + t.Fatal(err) + } + db, err := sql.Open(driverName, path) + if err != nil { + t.Fatal(err) + } + defer db.Close() + + if _, err := db.Exec(`CREATE TABLE student ( + so_bao_danh TEXT PRIMARY KEY, + ho_ten TEXT NOT NULL, + toan REAL + ); CREATE INDEX idx_ho_ten ON student(ho_ten);`); err != nil { + t.Fatal(err) + } + for _, r := range rows { + if _, err := db.Exec("INSERT INTO student VALUES (?,?,?)", r[0], r[1], r[2]); err != nil { + t.Fatal(err) + } + } +} + +var base = [][3]any{ + {"001", "Nguyễn Văn A", 8.0}, + {"002", "Trần Thị B", 7.25}, + {"003", "Lê Văn C", nil}, +} + +func compare(t *testing.T, a, b [][3]any) Result { + t.Helper() + dirA, dirB := t.TempDir(), t.TempDir() + makeDB(t, dirA, "x", a) + makeDB(t, dirB, "x", b) + + res, err := Compare([]registry.Dataset{{ID: "x"}}, dirA, dirB) + if err != nil { + t.Fatal(err) + } + return res[0] +} + +func TestIdenticalDatabasesMatch(t *testing.T) { + got := compare(t, base, base) + if !got.OK() { + t.Fatalf("identical databases reported problems: %v", got.Problems) + } + if got.HashA != got.HashB || got.HashA == "" { + t.Errorf("hashes should be equal and non-empty: %q %q", got.HashA, got.HashB) + } + if got.RowsA != 3 { + t.Errorf("RowsA = %d, want 3", got.RowsA) + } +} + +// TestOneChangedScoreIsCaught is the whole point: a single altered field, with +// the row count and every non-NULL count unchanged, must still fail. +func TestOneChangedScoreIsCaught(t *testing.T) { + changed := [][3]any{{"001", "Nguyễn Văn A", 8.25}, base[1], base[2]} + got := compare(t, base, changed) + + if got.OK() { + t.Fatal("a changed score was not detected") + } + if got.RowsA != got.RowsB { + t.Error("row counts should still match — that is what makes this hard") + } + joined := strings.Join(got.Problems, "; ") + if !strings.Contains(joined, "full-table hash differs") { + t.Errorf("problems = %v", got.Problems) + } + // The diagnosis must name the row and the column. + diff := strings.Join(got.FirstDiff, "\n") + if !strings.Contains(diff, "so_bao_danh=001") || !strings.Contains(diff, "toan") { + t.Errorf("FirstDiff should locate the change, got: %v", got.FirstDiff) + } +} + +// TestNullVersusValueIsCaught: a value becoming NULL changes the per-column +// count as well, so both signals should fire. +func TestNullVersusValueIsCaught(t *testing.T) { + nulled := [][3]any{base[0], base[1], {"003", "Lê Văn C", 5.0}} + got := compare(t, base, nulled) + if got.OK() { + t.Fatal("a NULL becoming a value was not detected") + } + if !strings.Contains(strings.Join(got.Problems, ";"), "non-NULL count for toan") { + t.Errorf("expected a per-column count difference, got %v", got.Problems) + } +} + +func TestRowCountDifferenceIsCaught(t *testing.T) { + got := compare(t, base, base[:2]) + if got.OK() { + t.Fatal("a missing row was not detected") + } + if !strings.Contains(strings.Join(got.Problems, ";"), "row count 3 vs 2") { + t.Errorf("problems = %v", got.Problems) + } +} + +// TestTextShiftAcrossFieldsIsCaught guards the field separator. Hashing the +// values joined bare would make ("ab","c") and ("a","bc") identical, so a value +// shifted across a column boundary would slip through. +func TestTextShiftAcrossFieldsIsCaught(t *testing.T) { + a := [][3]any{{"001", "ab", 1.0}} + b := [][3]any{{"001", "a", 1.0}} + // Same idea at the boundary between so_bao_danh and ho_ten. + c := [][3]any{{"01", "23", 1.0}} + d := [][3]any{{"012", "3", 1.0}} + + if compare(t, a, b).OK() { + t.Error("a shortened text value was not detected") + } + if compare(t, c, d).OK() { + t.Error("a value shifted across a field boundary was not detected") + } +} + +func TestMissingDatabaseIsAnError(t *testing.T) { + dir := t.TempDir() + makeDB(t, dir, "x", base) + if _, err := Compare([]registry.Dataset{{ID: "x"}}, dir, t.TempDir()); err == nil { + t.Fatal("comparing against a directory with no database must fail") + } +} + +func TestSer(t *testing.T) { + for _, tc := range []struct { + in any + want string + }{ + {nil, "\x00NULL"}, + {int64(7), "7"}, + {8.0, "8.0"}, // whole REALs get a fixed rendering + {7.25, "7.25"}, + {"x", "x"}, + {[]byte("y"), "y"}, + } { + if got := ser(tc.in); got != tc.want { + t.Errorf("ser(%v) = %q, want %q", tc.in, got, tc.want) + } + } + // A NULL must not collide with any real text value. + if ser(nil) == ser("NULL") || ser(nil) == ser("") { + t.Error("the NULL sentinel collides with a real value") + } +} diff --git a/crawler/internal/sources/sources_test.go b/crawler/internal/sources/sources_test.go index 593ea18..59fc24b 100644 --- a/crawler/internal/sources/sources_test.go +++ b/crawler/internal/sources/sources_test.go @@ -52,7 +52,7 @@ func resolveFixture(t *testing.T, src Source) []File { // parser sorts its inputs and inserts last-wins, so filenames decide which // row survives a duplicate exam number. If extraction or naming drifted, a // re-crawl could rebuild a database with the same row count and different -// content, which the row-count guard in build-db.js would not catch. +// content, which the assembler's row-count guard would not catch. // // The committed data// is the oracle: reading the source article and // applying the source's naming rule must reproduce it exactly, both directions. diff --git a/docs/data-pipeline.md b/docs/data-pipeline.md index 4ee55db..e003ba6 100644 --- a/docs/data-pipeline.md +++ b/docs/data-pipeline.md @@ -133,10 +133,22 @@ the 2017 configs listed 14. Neither list was complete: candidates could sit German, Japanese and Russian in both years, so 1,691 students ended up with **no foreign-language score at all**. -Running all 16 patterns everywhere recovered them — 182 Russian in 2016, and -German and Japanese across all three 2017 generations. See -`plans/reports/parser-parity-result.md` for the evidence that these are real -scores rather than false matches. +Running all 16 patterns everywhere recovered them: 182 × `tieng_nga` in 2016, +and 93 × `tieng_duc` plus 512 × `tieng_nhat` in 2017. + +Four things established that these are real scores and not false regex matches, +checked when the change was made: + +- Every student holds zero or exactly one foreign language, never two, so + nothing was double-counted. +- Each affected student had **all** language columns NULL beforehand — e.g. SBD + `01003198` went from all-NULL to `tieng_duc = 8`. +- The counts tracked dataset size across the three 2017 publications that then + existed (93/85/22 German, 512/484/313 Japanese). +- Every recovered value is an ordinary exam score in the 0–10 range. + +Row counts and the non-NULL counts of all 18 pre-existing columns were unchanged +across every dataset; only the four language columns gained values. ## Per-dataset quirks @@ -166,19 +178,26 @@ figure in the table above, and each `.db.gz` must be at least 90% of its usual size, or the build fails rather than publishing. That guard is the reason a truncated dataset cannot reach the site with a green pipeline. -For a deeper check, `parser/scripts/differential-parity.mjs` compares two sets -of databases field-by-field — row counts, per-column non-NULL counts, a -full-table SHA-256 over every row ordered by `so_bao_danh`, schema metadata, and -build stdout: +For a deeper check, `assemble verify` compares two sets of built databases +field by field — row counts, per-column non-NULL counts, schema metadata, and a +full-table SHA-256 over every row ordered by `so_bao_danh`: ```bash -node parser/scripts/differential-parity.mjs \ - --rust /path/to/a-{id}.db --go /path/to/b-{id}.db +cp -r .build/public/db /tmp/before # keep the databases you have +go -C assembler run ./cmd/assemble db # rebuild +go -C assembler run ./cmd/assemble verify /tmp/before .build/public/db ``` -It exits non-zero on any mismatch and fails loudly if a dataset is missing rather -than skipping it. Written for the Rust-to-Go migration, it works for any two -builds. Uses the built-in `node:sqlite`, so it needs no dependencies. +Each side is a directory of `.db.gz` (or `.db`); compressed databases are +expanded to a temporary file automatically. It exits non-zero on any mismatch, +names the first differing rows and columns, and fails rather than skipping when a +dataset is absent from either side — silently comparing one of two datasets is +how a gate passes without proving anything. + +This is the only check on database *content*. The reader oracle below covers +reading the spreadsheets and the row-count guard covers how many rows came out, +but a change in transform or writer logic can alter what is in those rows while +both of those still pass. `parser/internal/reader` additionally carries a frozen oracle of per-file cell-dump hashes covering all 182 inputs; `go -C parser test ./...` fails if any single @@ -194,8 +213,8 @@ go -C assembler run ./cmd/assemble db 2017 The row-count guard in `build:db` confirms the rebuild matches the expected total. That guard checks the count only, so if the crawl was expected to change -the data, compare content with `differential-parity.mjs` against a copy of the -previous database rather than trusting the count. +the data, compare content with `assemble verify` against a copy of the previous +database rather than trusting the count. ## Removed scripts @@ -203,8 +222,12 @@ previous database rather than trusting the count. were dropped with the Rust parser. The first two had been broken since before the repo was unified (a hardcoded Windows path in one, an undeclared `better-sqlite3` dependency in the other) and neither had any automated caller. -The latter two are superseded by `differential-parity.mjs`, which compares more -and cannot silently skip a dataset. +The latter two are superseded by `assemble verify`, which compares more and +cannot silently skip a dataset. `differential-parity.mjs`, which was that +comparator, has itself been folded into the assembler as +`assembler/internal/verify` — the pipeline is Go outside `web/`, and the port +also fixed a weakness: the JavaScript version hashed each row's fields joined +bare, so a value shifted across a column boundary produced the same digest. `crawl-baotintuc.js` was not dropped but rewritten as the Go `crawler/` module, producing the same local filenames. Two changes beyond the port: diff --git a/docs/deployment-guide.md b/docs/deployment-guide.md index 3f4cf4d..5b672b1 100644 --- a/docs/deployment-guide.md +++ b/docs/deployment-guide.md @@ -8,7 +8,7 @@ One-time setup: **Settings → Pages → Source: GitHub Actions**. ## What the workflow does 1. Checkout, Go toolchain, Node 24, `npm ci` in `web/` -2. Parser and crawler test suites, web lint, `govulncheck` over all three modules +2. Parser, crawler and assembler test suites, web lint, `govulncheck` over all three modules 3. `go -C assembler run ./cmd/assemble` — the whole pipeline: compile the parser, build and verify each database, compress it into `.build/public/db/`, run the Vite build, assemble `_site/` @@ -85,7 +85,7 @@ artifact — one missing line away from publishing it. - **No server-side compression is assumed.** The app fetches the `.gz` bytes directly rather than relying on `Content-Encoding: gzip`; Pages does not reliably compress arbitrary paths on the fly. -- **Total artifact is about 177 MB**, well inside the 1 GB site limit. +- **Total artifact is about 93 MB**, well inside the 1 GB site limit. ## Rollback diff --git a/parser/README.md b/parser/README.md index 519afd2..dbe8483 100644 --- a/parser/README.md +++ b/parser/README.md @@ -49,7 +49,7 @@ containing it. The port was gated on a field-by-field comparison of both implementations across the four datasets that existed then — 3,265,641 rows with identical full-table SHA-256, identical per-column non-NULL counts, identical schema metadata and -identical build stdout. `scripts/differential-parity.mjs` is that comparator and +identical build stdout. `assemble verify` is that comparator today and still runs against any two sets of databases. Behaviour was matched bug-for-bug, deliberately. Several quirks look like @@ -70,7 +70,14 @@ Each has a test naming it, so none can be tidied away by accident. `testdata/reader-fidelity-hashes.tsv` holds a SHA-256 per input file over a canonical dump of every cell of every sheet. It is **frozen**: it was produced by the Rust reader, which no longer exists, so it cannot be regenerated. It -still fails if any single cell of any of the 182 files reads differently. +still fails if any single cell of any input file reads differently. + +A mismatch names the file but not the cell. `cmd/dumpcells` prints the stream +the hash is taken over, so two runs can be diffed: + +```bash +go -C parser run ./cmd/dumpcells ../data/2017/an-giang.xls out.tsv +``` The assembler refuses to publish a database whose row count does not match the known figure, or whose artifact is under 90% of its usual size. diff --git a/parser/cmd/dumpcells/main.go b/parser/cmd/dumpcells/main.go index e209b49..9430adc 100644 --- a/parser/cmd/dumpcells/main.go +++ b/parser/cmd/dumpcells/main.go @@ -1,6 +1,11 @@ -// Command dumpcells emits the canonical cell rendering of a spreadsheet, for -// comparison against the Rust/calamine ground truth produced by -// parser/examples/dump_cells.rs. +// Command dumpcells emits the canonical cell rendering of a spreadsheet. +// +// It exists to diagnose a reader-fidelity failure. That suite compares a +// SHA-256 per input file against a frozen oracle, so a mismatch says which file +// changed but not which cell; this prints the stream the hash is taken over, so +// two runs can be diffed. It was originally written to compare against the +// Rust/calamine ground truth from parser/examples/dump_cells.rs, which was +// removed with the Rust tree — the hashes it produced are what remains. // // The canonical stream carries geometry and rendered cell values only. The // calamine Data variant is deliberately excluded: Data::Empty and diff --git a/parser/cmd/xlsxread/main.go b/parser/cmd/xlsxread/main.go index cf7dc89..332c899 100644 --- a/parser/cmd/xlsxread/main.go +++ b/parser/cmd/xlsxread/main.go @@ -1,8 +1,8 @@ // Command xlsxread reads .xls/.xlsx files and builds SQLite databases for the // thptqg datasets. // -// The CLI contract is fixed by parser/scripts/build-db.js and must match the -// Rust binary exactly: +// The CLI contract is fixed by the assembler, which drives this binary, and it +// matched the Rust binary it replaced exactly: // // xlsxread build --schema --input --output // xlsxread audit --schema --input --db diff --git a/parser/internal/config/config.go b/parser/internal/config/config.go index 2860990..e4749d1 100644 --- a/parser/internal/config/config.go +++ b/parser/internal/config/config.go @@ -2,8 +2,8 @@ // // The config deliberately carries no SQL. The table shape, the INSERT and the // subject regexes are identical for every dataset and live in internal/schema; -// keeping them here meant four copies of the same DDL, which is how the 2016 and -// 2017 schemas drifted apart. +// keeping them here meant one copy of the DDL per dataset, which is how the +// 2016 and 2017 schemas drifted apart. package config import ( diff --git a/parser/internal/ingest/detect2016.go b/parser/internal/ingest/detect2016.go index 1cf43a2..371f128 100644 --- a/parser/internal/ingest/detect2016.go +++ b/parser/internal/ingest/detect2016.go @@ -309,7 +309,7 @@ func Detect2016(cfg *config.DatasetConfig, inputDir, outputPath string) error { if err := tx.Commit(); err != nil { return fmt.Errorf("commit: %w", err) } - return writer.Finish(db, outputPath, st, label, false) + return writer.Finish(db, outputPath, st) } // processFile2016 detects the layout per SHEET and processes that sheet's rows. diff --git a/parser/internal/ingest/ingest.go b/parser/internal/ingest/ingest.go index 377a1d0..5da024f 100644 --- a/parser/internal/ingest/ingest.go +++ b/parser/internal/ingest/ingest.go @@ -147,7 +147,6 @@ func Standard(cfg *config.DatasetConfig, inputDir, outputPath string) error { } defer db.Close() - isOld2 := strings.Contains(label, "old2") stripBlank := cfg.Reader.StripBlankRows // One transaction spans the whole dataset directory (main.rs:120,184). @@ -230,7 +229,7 @@ func Standard(cfg *config.DatasetConfig, inputDir, outputPath string) error { } // VACUUM only after COMMIT — SQLite refuses it inside a transaction. - return writer.Finish(db, outputPath, st, label, isOld2) + return writer.Finish(db, outputPath, st) } // cellAt returns the trimmed cell at idx, or "" when out of range — the diff --git a/parser/internal/reader/fidelity_test.go b/parser/internal/reader/fidelity_test.go index 133f096..44851ae 100644 --- a/parser/internal/reader/fidelity_test.go +++ b/parser/internal/reader/fidelity_test.go @@ -30,7 +30,7 @@ import ( // under -short, which is how to iterate without paying ~77s to re-read 418 MB. func TestReaderFidelity(t *testing.T) { if testing.Short() { - t.Skip("-short: skipping the 299-file corpus sweep") + t.Skip("-short: skipping the full-corpus sweep") } root := repoRoot(t) manifest := filepath.Join(root, "parser", "testdata", "reader-fidelity-hashes.tsv") diff --git a/parser/internal/reader/reader.go b/parser/internal/reader/reader.go index 79f3fda..fd1a1eb 100644 --- a/parser/internal/reader/reader.go +++ b/parser/internal/reader/reader.go @@ -2,9 +2,12 @@ // contract, mirroring parser/src/reader.rs (which wraps calamine's // open_workbook_auto for both formats). // -// The contract is deliberately row-streaming rather than whole-workbook -// materialising: data/2017/ha-noi.xls alone holds 72k rows across two sheets, -// and the Rust original keeps at most one sheet range live at a time. +// The *contract* is row-at-a-time: callers receive one row per callback and +// never hold the workbook. The two implementations do not honour that in +// memory — both decode the whole workbook into [][][]Cell up front, because +// neither underlying library exposes a usable row cursor. data/2017/ha-noi.xls +// alone holds 72k rows across two sheets, so a caller that assumed streaming +// memory here would be wrong. package reader import ( @@ -27,7 +30,7 @@ type Cell struct { // Sheet carries a sheet's identity and geometry. Height and Width describe the // used range, matching calamine's Range::height()/width(); every used range in -// the corpus starts at (0,0), verified across all 299 files. +// the corpus starts at (0,0), verified across every input file. type Sheet struct { Index int Name string diff --git a/parser/internal/schema/schema.go b/parser/internal/schema/schema.go index 3e7acc7..fd6487e 100644 --- a/parser/internal/schema/schema.go +++ b/parser/internal/schema/schema.go @@ -1,8 +1,7 @@ // Package schema is the single source of truth for the SQL shape of every // dataset — a direct port of parser/src/schema.rs. // -// All four datasets (2016, 2017, 2017-old, 2017-old2) write into the same -// 22-column student table. Columns a dataset has no data for bind NULL. +// Every dataset writes into the same 22-column student table. Columns a dataset has no data for bind NULL. // // Column provenance: // diff --git a/parser/internal/writer/writer.go b/parser/internal/writer/writer.go index a820759..a82ad07 100644 --- a/parser/internal/writer/writer.go +++ b/parser/internal/writer/writer.go @@ -117,12 +117,12 @@ type Stats struct { // // VACUUM must run AFTER the transaction commits — SQLite refuses it inside one. // -// The wording branches on datasetLabel, which Rust derives from the input -// directory's basename (main.rs:98-101). That makes the output depend on a -// filesystem path rather than on config; it is reproduced here for parity, and -// the caller passes the label explicitly so tests are not at the mercy of a -// temp-directory name. -func Finish(db *sql.DB, dbPath string, st Stats, datasetLabel string, isOld2 bool) error { +// The Rust original varied the wording by dataset, keying off the input +// directory's basename (main.rs:98-101) — one phrasing for the 2017-old2 +// export, another for anything with "old" in its name. Both of those datasets +// have been removed, so only one branch was ever taken; the wording below is +// exactly what 2016 and 2017 have always printed. +func Finish(db *sql.DB, dbPath string, st Stats) error { if _, err := db.Exec("VACUUM"); err != nil { return fmt.Errorf("vacuum: %w", err) } @@ -135,22 +135,13 @@ func Finish(db *sql.DB, dbPath string, st Stats, datasetLabel string, isOld2 boo insertable := st.SourceRows - st.Skipped fmt.Println() - if isOld2 { - fmt.Printf("Source non-blank data rows: %d\n", st.SourceRows) - fmt.Printf(" skipped (empty/non-numeric SBD): %d\n", st.Skipped) - } else { - fmt.Printf("Source data rows (post-header): %d\n", st.SourceRows) - if containsOld(datasetLabel) { - fmt.Printf(" skipped (empty/non-numeric SBD): %d\n", st.Skipped) - } else { - fmt.Printf(" skipped (empty/invalid): %d\n", st.Skipped) - } - } + fmt.Printf("Source data rows (post-header): %d\n", st.SourceRows) + fmt.Printf(" skipped (empty/invalid): %d\n", st.Skipped) fmt.Printf(" insertable: %d\n", insertable) fmt.Printf(" insert errors: %d\n", st.Errors) fmt.Printf("DB rows (distinct SBD): %d\n", dbCount) - if !containsOld(datasetLabel) && st.Errors == 0 { + if st.Errors == 0 { gap := int64(insertable) - dbCount if gap == 0 { fmt.Println("Audit: OK — every source row made it in.") @@ -166,12 +157,3 @@ func Finish(db *sql.DB, dbPath string, st Stats, datasetLabel string, isOld2 boo fmt.Printf("Size: %.1f MB\n", float64(size)/1024.0/1024.0) return nil } - -func containsOld(label string) bool { - for i := 0; i+3 <= len(label); i++ { - if label[i:i+3] == "old" { - return true - } - } - return false -} diff --git a/parser/scripts/differential-parity.mjs b/parser/scripts/differential-parity.mjs deleted file mode 100644 index 18d6e184c72570932f14419d180964bc3c386441..0000000000000000000000000000000000000000 GIT binary patch literal 0 HcmV?d00001 literal 7594 zcmb_h?QYvf7Tw*Seu`T*XeGy@6SvzIGTg%TN9(O)XDw%eA_*dj98ptKB+DUftFD0k zwGXh^C+w5#IWr_h*>0Kz7HDIMoVjy9&b{~C(Z?TcSEksW=yY49*OF$D+SuNfLp`0U zLZzh+63K%?mkWu5QVk_9vTJ3eoGJM(<13?zPmM%D8B79$H$kB!$zmM_NwT0U^>d}K zg9O5)%;O$}*}nK6X>U{^a;Q$j%6;uj}xSCV& zG%M`uxsHpV)LH7wt5lZ~q>;o~5~);5WlBSFmBIc+3TdSdWvXw>s!(PZ`&Y+gX|j-T zrot-;;y|a+m+F1~{406!tz=UvRFF`ZPV=g46g^alo+v1wu>0)%QYMv7O3BjN*liOQ zIxo%kwJP*sFce=Y4B&u_FrYIhSvl(fU=~~}8w?G%!^J?*&Q5bWqK%M1Nm{x8^5w<( zKaLLHT^t?Y@5PT2S3wcUO;#n5s5t^?)HTdAc~xi}Ez>|Jj23+wNUn;h3QGqw7+v6t z9Ta0@M(cMiYOC%FPVof(Bx=YOamZX-rSftto9-5tdU$V+f=qOVutv6^I4^ z0dFKaYJmsrlv!botsP&=O<=9OLH3k56&P)R@`^e!*M{cjM3)L)Nr0$mVf|L9JVQ!Z z437SOiuV#GnlS(EpZ~!&VKgI^s}v3n7h6=AUd0LQi528kS&BIlDY+J)TspxWamE_dE<7831gRk6TFBs z_ZU)}hwZl4Zr{L^QC2$(pc)j;V47-TN?lf^iinOT&b}PoGBMID z9sHt-%)%#KF!uAH3}^nvhMp5Rawj1IQL66)Gn1tQg=Zrb7K^;hyxrAGb&H^z(IO4E zSP!COfQ;YP_f1WA&tY9BAFgwu!`h87W4$@tzuX@kU5;c=UVA&wo_~Q<sK_XJRhFIi!sEqNk^4iw`)9v}xZt3r0 zU7=7HQ!M;?PoBvqpR6r^t>vjae=FZS`gGVHxx2A!>>)B<2FL^m-Lu+x4gj>_0d_p- z;53c12fN~|-P6D!nyhy4r1R?W!)h$G`&W2%fLck3v&Uq#Xlx}k8rn=R*pK%$CeeY{fJS{k$@>-uTH~WR^pD za)EI&V^>A*PFUv9M5exMZm&>OH;rC2 zapeFN`B698J| z!*q5_t>W%uM|=3X@I!=d$Pcu{eJU4u{VC1Lv`P~9IKM(XoRFU~o%Xz6iLW*XquclB zW)t!qdvM2}f2i9^r7G1wSG7`bAemgNj6tdEJ*HS4UYJSGE#IV1fYaOMxCz&WY|04a z#RkYW6|}($e995K8LKACume<0?4m-0@PN+ACih>kSui!gAJu$UPKFs9}IS z1Wy*F8mzd)6jj!;Ov+-_M*AjuCm?JrZAE44-!57`D3v8Uw#HL0V!h_3T#T*^s$e-# z48BQ$AVEmnM2D7pfWD+E!8S-jcO~g#--p2=7Q#iK07`CjIdi<`;%NW0)eZC+SqAg? z9zX`u#FwIiu1G`lX39;injkrar~ov}`HF;EO@k^asS}6;EPyJpfCQrL!w|Sd(V?AG zMbGYCyDZgH>0Qg--kuPP^6RhSzE#j@3X@Qc>PUNkqWS1LBHTnAgWbB-URohMh1V`V zv)_GfPT;!O9ruk~`(<`a+j`IYGL(@`QfSg?BAsV0?P=Bp-3OaS-|r80{`!Sru7DHp z&q9jqX3G^RcnpEf3q23;{7Nkv6hWjy`oW5Qdb zBvHU7WIW>+D%*3V4~EkUp5TK*aw_TdopvHKzsgZ=71k!kmFLiwXMzYjUx#n|ec#or zKYM+9P-mC-7iR=GiF3qt}YBH1GFpBL7ZkrgO+z;Kc$dnUIm}&6%0k(#A8bm%|T*McAj-( zL(1v;Y*UU4jR)A=hQj;sP@vPbKBhlh2x3ZvCjSqKHnO|BO#tf{Q)BI?>K5gk=e8&F z;Pn?w@4}Bb2%sLs_ETRb$f7H^vxA+G$XTJ8a=G1t!?y*H0FjSSVE8OvJ!|yi7P#0k z6ss7Yh>WF+;5~LxE6?xXdI>5Mg3yqgVa4A@=Vw0ZI%)-mX4udwu*9_Ta6w0#<+we% zb`N#v*zAzo{7W0DG-kyZB^$)oVyEM!4$iT#1#&&CII5yemV(#;t5g|H5@RPD4>sVg z023)hgOBmi3{wM&9w+y8P}lSS$c{$feqH1iIfCohWpAD8ZV6U);MZB$WwY%v+WXOJ zis==3pV;U=iu1N+5OP_fd^FG=!|28~IuupkTxoQs2*oGJ$-_`Gc`Jj4(|2X6sSi|- zUUUE^*@HDb9K83nja7G$B6Ieta{3k0t0B(Yzy0gq@@DjO8?X5B@LN}kXbYuTBW&m6 zmk(9Hr58@wrKkkO=+0_Hwr%ORSDWnUifKyVd;PKrihFo+e0+2<5`3=3H&apLPLB$= zENFG-Hgsg^+K?+@&6|{mZges=!ac#XofY2)lX{A3gYN8Z>>RFm@q4J3CuK?GIidhv6EQN3Wao;56nF@4TaafSjStYq0O2aoTIe{bLFKulLS@&2R*J-X5#Uxe;w7tuupm#}BV_|P

zn_#d>!Z-NCyH#=zml$=Z4lJ2`oK_s1zdF0@ZT2Nxqk5=y7)CHOhn-5^$a#!Re2^UL ze<#I8z<%s^AWFIzT5XX8_U%$ocb?qE?u)Bj_xwjn=bXGCGD#C+G^|GOOPbcgr`?iAkIJ2nLGHW9Hd;x?yhr%=fw zi)~tTUF<^X+Hli~J2F}gAb!YS0w|MXf}=&t2JG(fDn3Z;XRE-D?WX4KvW@Z4;kb!# z9WG8ut2qjhs|TqL{MVNtD4iwN$ur+xJy>YF&yB-Zo}FS89=MAMySQ%SjW_N=yR-UX zWqFg*_ccE)+=0Hqt%2%2+mg?A`nWma3e-lgl6Fc>m<|=+OQa^5pK`t|f^8O82~H z`a`E^$G=nS-1?k#C&EQ!H`6+`nyYb;;}X{zRI;mzSF3XVgH?k$@_F-HgK2J 2017= 2017-old= 2017-old2= > current.json -node parser/scripts/verify-parity.js \ - plans/reports/parser-parity-baseline.json current.json -``` - -The baseline cannot be regenerated — it required the two pre-refactor crates, -which no longer exist. It is committed for that reason. - -## Unresolved questions - -None. diff --git a/web/index.html b/web/index.html index 6961ca3..d3ecfeb 100644 --- a/web/index.html +++ b/web/index.html @@ -2,7 +2,6 @@ - @@ -10,7 +9,12 @@ rel="stylesheet" href="https://fonts.googleapis.com/css2?family=Be+Vietnam+Pro:wght@400;500;600;700&display=swap" /> - Tra cứu điểm thi THPT QG 2017 + + Tra cứu điểm thi THPT Quốc gia

diff --git a/web/src/App.css b/web/src/App.css index 376bd76..ef77bf7 100644 --- a/web/src/App.css +++ b/web/src/App.css @@ -14,7 +14,6 @@ --focus-ring: 0 0 0 3px rgba(59, 79, 228, 0.35); --radius: 10px; --radius-sm: 6px; - --shadow-sm: 0 1px 2px rgba(15, 20, 40, 0.05); --shadow-md: 0 6px 20px rgba(15, 20, 40, 0.08); /* Score tier palette — TFT rarity ladder (white → green → blue → purple → gold → prismatic). @@ -49,7 +48,6 @@ --primary-hover: #879aff; --primary-ink: #0f1220; --focus-ring: 0 0 0 3px rgba(109, 130, 255, 0.45); - --shadow-sm: 0 1px 2px rgba(0, 0, 0, 0.4); --shadow-md: 0 6px 20px rgba(0, 0, 0, 0.55); --tier-common-bg: #242838; --tier-common-ink: #c5cad8; --tier-common-accent: #4a5163; @@ -853,16 +851,7 @@ footer { font-size: 0.9rem; } -.hub-archive h2 { - font-size: 0.95rem; - font-weight: 600; - color: var(--text-muted); - margin-bottom: 0; -} -.hub-list-compact { - margin-top: 0.75rem; -} .hub-back { margin: 0.35rem 0 0; diff --git a/web/src/App.jsx b/web/src/App.jsx index 930a6c6..24e3556 100644 --- a/web/src/App.jsx +++ b/web/src/App.jsx @@ -147,6 +147,12 @@ function DatasetApp({ dataset }) { writeUrlQuery(""); }, []); + // The assembler copies one index.html to every route, so its static + // cannot name a dataset. Set the real one once the route is resolved. + useEffect(() => { + document.title = dataset.title; + }, [dataset.title]); + return ( <div className="app"> <header> diff --git a/web/src/assets/vite.svg b/web/src/assets/vite.svg deleted file mode 100644 index 5101b67..0000000 --- a/web/src/assets/vite.svg +++ /dev/null @@ -1 +0,0 @@ -<svg xmlns="http://www.w3.org/2000/svg" width="77" height="47" fill="none" aria-labelledby="vite-logo-title" viewBox="0 0 77 47"><title id="vite-logo-title">Vite diff --git a/web/src/components/hub.jsx b/web/src/components/hub.jsx index 9343759..69d9fb0 100644 --- a/web/src/components/hub.jsx +++ b/web/src/components/hub.jsx @@ -8,8 +8,6 @@ import { DATASETS, pathOf } from "../datasets"; * No database is fetched on this route. */ export function Hub() { - const current = DATASETS.filter((d) => !d.id.includes("old")); - const archived = DATASETS.filter((d) => d.id.includes("old")); const base = import.meta.env.BASE_URL; return ( @@ -24,7 +22,7 @@ export function Hub() {
- -
-

Phiên bản cũ của trang 2017

-
-