mirror of
https://github.com/tiennm99/thptqg.git
synced 2026-10-11 03:13:48 +00:00
Merge pull request #7 from tiennm99/chore/remove-js-and-dead-code
chore: remove the last JS script and the dead weight three audits found
This commit is contained in:
33 files changed
+749
-8814
No files matched your search
@@ -62,6 +62,12 @@ jobs:
|
||||
- name: Test crawler
|
||||
run: go -C crawler test ./...
|
||||
|
||||
# The assembler owns every guard between a built database and the
|
||||
# published site, so its tests are the ones that prove a short or missing
|
||||
# database cannot ship.
|
||||
- name: Test assembler
|
||||
run: go -C assembler test ./...
|
||||
|
||||
- name: Lint web
|
||||
working-directory: web
|
||||
run: npm run lint
|
||||
|
||||
@@ -59,6 +59,7 @@ is missing altogether. Sub-steps when iterating:
|
||||
```bash
|
||||
go -C assembler run ./cmd/assemble db 2017 # one database
|
||||
go -C assembler run ./cmd/assemble site # web build and _site only
|
||||
go -C assembler run ./cmd/assemble verify A B # compare two sets of databases
|
||||
(cd web && npm run dev) # the app against staged databases
|
||||
```
|
||||
|
||||
|
||||
@@ -4,6 +4,7 @@
|
||||
// assemble # databases, then the site
|
||||
// assemble db # databases only (add ids to limit: assemble db 2017)
|
||||
// assemble site # web build and _site only, reusing staged databases
|
||||
// assemble verify A B # compare two sets of built databases field by field
|
||||
//
|
||||
// It sequences the other stages rather than doing their work: the parser reads
|
||||
// spreadsheets, Vite bundles the app, and this decides what runs, checks what
|
||||
@@ -18,6 +19,7 @@ import (
|
||||
"github.com/tiennm99/thptqg/assembler/internal/databases"
|
||||
"github.com/tiennm99/thptqg/assembler/internal/registry"
|
||||
"github.com/tiennm99/thptqg/assembler/internal/site"
|
||||
"github.com/tiennm99/thptqg/assembler/internal/verify"
|
||||
)
|
||||
|
||||
func main() {
|
||||
@@ -34,9 +36,20 @@ Steps:
|
||||
(none) databases, then the site
|
||||
db build, verify and compress the databases
|
||||
site build the web app and assemble _site
|
||||
verify A B compare two directories of built databases, field by field
|
||||
|
||||
Naming datasets limits the db step to those; the site step always covers all of
|
||||
them, since a partial site would publish links to databases it did not build.
|
||||
|
||||
verify takes two directories holding <id>.db.gz (or <id>.db) — for example a
|
||||
copy of the previous .build/public/db against a fresh build. Relative paths
|
||||
resolve against the repository root, not the working directory. It reports row
|
||||
counts, per-column non-NULL counts, a full-table hash and the first differing
|
||||
rows.
|
||||
|
||||
Nothing else checks database content: the reader oracle covers reading and the
|
||||
row-count guard covers how many rows, but a change in transform or writer logic
|
||||
can alter what is in them while both still pass.
|
||||
`)
|
||||
}
|
||||
|
||||
@@ -44,7 +57,7 @@ func run(args []string) error {
|
||||
step := ""
|
||||
if len(args) > 0 {
|
||||
switch args[0] {
|
||||
case "db", "site":
|
||||
case "db", "site", "verify":
|
||||
step, args = args[0], args[1:]
|
||||
case "-h", "--help":
|
||||
usage()
|
||||
@@ -66,6 +79,19 @@ func run(args []string) error {
|
||||
return err
|
||||
}
|
||||
|
||||
if step == "verify" {
|
||||
if len(args) != 2 {
|
||||
usage()
|
||||
return fmt.Errorf("verify needs two directories")
|
||||
}
|
||||
// Relative paths resolve against the repository root, not the working
|
||||
// directory. The documented invocation is `go -C assembler run ...`,
|
||||
// and -C moves the process into assembler/ before main starts — so a
|
||||
// bare `.build/public/db` would otherwise look for it in there and
|
||||
// report a missing database that is sitting in plain sight.
|
||||
return runVerify(all, atRoot(root, args[0]), atRoot(root, args[1]))
|
||||
}
|
||||
|
||||
if step == "" || step == "db" {
|
||||
selected, err := registry.Select(all, args)
|
||||
if err != nil {
|
||||
@@ -88,6 +114,48 @@ func run(args []string) error {
|
||||
return nil
|
||||
}
|
||||
|
||||
// atRoot resolves a relative path against the repository root.
|
||||
func atRoot(root, path string) string {
|
||||
if filepath.IsAbs(path) {
|
||||
return path
|
||||
}
|
||||
return filepath.Join(root, path)
|
||||
}
|
||||
|
||||
// runVerify compares two sets of built databases and fails loudly on any
|
||||
// difference. A dataset missing from either side is an error rather than a
|
||||
// skip: silently comparing one of two datasets is how a gate passes without
|
||||
// proving anything.
|
||||
func runVerify(all []registry.Dataset, dirA, dirB string) error {
|
||||
results, err := verify.Compare(all, dirA, dirB)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
bad := 0
|
||||
for _, r := range results {
|
||||
fmt.Printf("--- %s ---\n", r.ID)
|
||||
fmt.Printf(" rows : %d vs %d\n", r.RowsA, r.RowsB)
|
||||
if r.OK() {
|
||||
fmt.Printf(" full-table : identical sha256 %s\n", r.HashA[:16])
|
||||
continue
|
||||
}
|
||||
bad++
|
||||
for _, p := range r.Problems {
|
||||
fmt.Printf(" *** %s\n", p)
|
||||
}
|
||||
for _, d := range r.FirstDiff {
|
||||
fmt.Println(d)
|
||||
}
|
||||
}
|
||||
|
||||
if bad > 0 {
|
||||
return fmt.Errorf("%d dataset(s) differ", bad)
|
||||
}
|
||||
fmt.Println("\nIdentical — rows, per-column non-NULL counts, schema and full-table hash.")
|
||||
return nil
|
||||
}
|
||||
|
||||
func buildDatabases(root string, all, selected []registry.Dataset) error {
|
||||
p := databases.DefaultPaths(root)
|
||||
|
||||
|
||||
@@ -26,7 +26,6 @@ import (
|
||||
|
||||
// Paths locates the pieces this package needs.
|
||||
type Paths struct {
|
||||
Root string
|
||||
// Web is the Vite project directory.
|
||||
Web string
|
||||
// Dist is where Vite emits, inside the web workspace.
|
||||
@@ -39,7 +38,6 @@ type Paths struct {
|
||||
func DefaultPaths(root string) Paths {
|
||||
web := filepath.Join(root, "web")
|
||||
return Paths{
|
||||
Root: root,
|
||||
Web: web,
|
||||
Dist: filepath.Join(web, "dist"),
|
||||
Site: filepath.Join(root, "_site"),
|
||||
|
||||
@@ -22,7 +22,7 @@ func fakeBuild(t *testing.T, dbs ...string) Paths {
|
||||
for _, name := range dbs {
|
||||
write(t, filepath.Join(dist, "db", name), "gzipped-bytes")
|
||||
}
|
||||
return Paths{Root: root, Web: filepath.Join(root, "web"), Dist: dist, Site: filepath.Join(root, "_site")}
|
||||
return Paths{Web: filepath.Join(root, "web"), Dist: dist, Site: filepath.Join(root, "_site")}
|
||||
}
|
||||
|
||||
func write(t *testing.T, path, body string) {
|
||||
@@ -113,7 +113,7 @@ func TestGzipIsNotMistakenForRaw(t *testing.T) {
|
||||
|
||||
func TestAssembleRejectsAMissingBuild(t *testing.T) {
|
||||
root := t.TempDir()
|
||||
p := Paths{Root: root, Web: root, Dist: filepath.Join(root, "dist"), Site: filepath.Join(root, "_site")}
|
||||
p := Paths{Web: root, Dist: filepath.Join(root, "dist"), Site: filepath.Join(root, "_site")}
|
||||
if err := Assemble(p, datasets); err == nil {
|
||||
t.Fatal("expected an error when there is no Vite build")
|
||||
}
|
||||
|
||||
@@ -0,0 +1,384 @@
|
||||
// Package verify compares two sets of built databases field by field.
|
||||
//
|
||||
// Nothing else checks database *content*. The reader-fidelity oracle covers
|
||||
// reading the spreadsheets, and the row-count guard covers how many rows came
|
||||
// out — but a change in transform or writer logic can alter what is in those
|
||||
// rows while both of those still pass. This is what catches that.
|
||||
//
|
||||
// It was written as the gate for the Rust-to-Go parser migration and works for
|
||||
// any two builds: before and after a refactor, or a re-crawl against the
|
||||
// databases already published.
|
||||
package verify
|
||||
|
||||
import (
|
||||
"compress/gzip"
|
||||
"crypto/sha256"
|
||||
"database/sql"
|
||||
"encoding/hex"
|
||||
"fmt"
|
||||
"io"
|
||||
"math"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"sort"
|
||||
"strconv"
|
||||
"strings"
|
||||
|
||||
_ "modernc.org/sqlite"
|
||||
|
||||
"github.com/tiennm99/thptqg/assembler/internal/registry"
|
||||
)
|
||||
|
||||
const driverName = "sqlite"
|
||||
|
||||
// Result is the outcome for one dataset.
|
||||
type Result struct {
|
||||
ID string
|
||||
Problems []string
|
||||
RowsA int64
|
||||
RowsB int64
|
||||
HashA string
|
||||
HashB string
|
||||
FirstDiff []string
|
||||
}
|
||||
|
||||
// OK reports whether the two databases are logically identical.
|
||||
func (r Result) OK() bool { return len(r.Problems) == 0 }
|
||||
|
||||
// Compare checks every dataset in the registry, reading <dir>/<id>.db.gz (or
|
||||
// <id>.db) from each side.
|
||||
func Compare(datasets []registry.Dataset, dirA, dirB string) ([]Result, error) {
|
||||
out := make([]Result, 0, len(datasets))
|
||||
for _, d := range datasets {
|
||||
r, err := compareOne(d.ID, dirA, dirB)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
out = append(out, r)
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
func compareOne(id, dirA, dirB string) (Result, error) {
|
||||
res := Result{ID: id}
|
||||
|
||||
a, cleanA, err := open(dirA, id)
|
||||
if err != nil {
|
||||
return res, err
|
||||
}
|
||||
defer cleanA()
|
||||
defer a.Close()
|
||||
|
||||
b, cleanB, err := open(dirB, id)
|
||||
if err != nil {
|
||||
return res, err
|
||||
}
|
||||
defer cleanB()
|
||||
defer b.Close()
|
||||
|
||||
sigA, err := schemaSignature(a)
|
||||
if err != nil {
|
||||
return res, err
|
||||
}
|
||||
sigB, err := schemaSignature(b)
|
||||
if err != nil {
|
||||
return res, err
|
||||
}
|
||||
if sigA != sigB {
|
||||
res.Problems = append(res.Problems, "schema/index metadata differs")
|
||||
}
|
||||
|
||||
cols, err := columns(a)
|
||||
if err != nil {
|
||||
return res, err
|
||||
}
|
||||
|
||||
if res.RowsA, err = count(a, "SELECT COUNT(*) FROM student"); err != nil {
|
||||
return res, err
|
||||
}
|
||||
if res.RowsB, err = count(b, "SELECT COUNT(*) FROM student"); err != nil {
|
||||
return res, err
|
||||
}
|
||||
if res.RowsA != res.RowsB {
|
||||
res.Problems = append(res.Problems, fmt.Sprintf("row count %d vs %d", res.RowsA, res.RowsB))
|
||||
}
|
||||
|
||||
// Per-column non-NULL counts localise a difference to a column even when the
|
||||
// full-table hash has already said "something differs".
|
||||
for _, c := range cols {
|
||||
q := fmt.Sprintf("SELECT COUNT(%s) FROM student", c)
|
||||
na, err := count(a, q)
|
||||
if err != nil {
|
||||
return res, err
|
||||
}
|
||||
nb, err := count(b, q)
|
||||
if err != nil {
|
||||
return res, err
|
||||
}
|
||||
if na != nb {
|
||||
res.Problems = append(res.Problems, fmt.Sprintf("non-NULL count for %s: %d vs %d", c, na, nb))
|
||||
}
|
||||
}
|
||||
|
||||
if res.HashA, err = tableHash(a, cols); err != nil {
|
||||
return res, err
|
||||
}
|
||||
if res.HashB, err = tableHash(b, cols); err != nil {
|
||||
return res, err
|
||||
}
|
||||
if res.HashA != res.HashB {
|
||||
res.Problems = append(res.Problems, "full-table hash differs")
|
||||
if res.FirstDiff, err = firstDifferences(a, b, cols, 20); err != nil {
|
||||
return res, err
|
||||
}
|
||||
}
|
||||
|
||||
return res, nil
|
||||
}
|
||||
|
||||
// open finds <id>.db.gz or <id>.db in dir and returns a handle. A compressed
|
||||
// database is expanded to a temporary file, since SQLite needs to seek.
|
||||
func open(dir, id string) (*sql.DB, func(), error) {
|
||||
noop := func() {}
|
||||
|
||||
plain := filepath.Join(dir, id+".db")
|
||||
if _, err := os.Stat(plain); err == nil {
|
||||
db, err := sql.Open(driverName, "file:"+plain+"?mode=ro")
|
||||
return db, noop, err
|
||||
}
|
||||
|
||||
gzPath := filepath.Join(dir, id+".db.gz")
|
||||
f, err := os.Open(gzPath)
|
||||
if err != nil {
|
||||
return nil, noop, fmt.Errorf("no %s.db or %s.db.gz in %s", id, id, dir)
|
||||
}
|
||||
defer f.Close()
|
||||
|
||||
zr, err := gzip.NewReader(f)
|
||||
if err != nil {
|
||||
return nil, noop, fmt.Errorf("%s: %w", gzPath, err)
|
||||
}
|
||||
defer zr.Close()
|
||||
|
||||
tmp, err := os.CreateTemp("", "verify-"+id+"-*.db")
|
||||
if err != nil {
|
||||
return nil, noop, err
|
||||
}
|
||||
cleanup := func() { os.Remove(tmp.Name()) }
|
||||
|
||||
if _, err := io.Copy(tmp, zr); err != nil {
|
||||
tmp.Close()
|
||||
cleanup()
|
||||
return nil, noop, err
|
||||
}
|
||||
if err := tmp.Close(); err != nil {
|
||||
cleanup()
|
||||
return nil, noop, err
|
||||
}
|
||||
|
||||
db, err := sql.Open(driverName, "file:"+tmp.Name()+"?mode=ro")
|
||||
if err != nil {
|
||||
cleanup()
|
||||
return nil, noop, err
|
||||
}
|
||||
return db, cleanup, nil
|
||||
}
|
||||
|
||||
func count(db *sql.DB, query string) (int64, error) {
|
||||
var n int64
|
||||
err := db.QueryRow(query).Scan(&n)
|
||||
return n, err
|
||||
}
|
||||
|
||||
// columns returns the column names in declaration order.
|
||||
func columns(db *sql.DB) ([]string, error) {
|
||||
rows, err := db.Query("SELECT name FROM pragma_table_info('student') ORDER BY cid")
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer rows.Close()
|
||||
|
||||
var out []string
|
||||
for rows.Next() {
|
||||
var name string
|
||||
if err := rows.Scan(&name); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
out = append(out, name)
|
||||
}
|
||||
if err := rows.Err(); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if len(out) == 0 {
|
||||
return nil, fmt.Errorf("no student table")
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// schemaSignature reduces the table and index metadata to a comparable string.
|
||||
func schemaSignature(db *sql.DB) (string, error) {
|
||||
rows, err := db.Query("SELECT cid, name, type, \"notnull\", pk FROM pragma_table_info('student') ORDER BY cid")
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
var cols []string
|
||||
for rows.Next() {
|
||||
var cid, notnull, pk int
|
||||
var name, typ string
|
||||
if err := rows.Scan(&cid, &name, &typ, ¬null, &pk); err != nil {
|
||||
rows.Close()
|
||||
return "", err
|
||||
}
|
||||
cols = append(cols, fmt.Sprintf("%d:%s:%s:%d:%d", cid, name, typ, notnull, pk))
|
||||
}
|
||||
rows.Close()
|
||||
if err := rows.Err(); err != nil {
|
||||
return "", err
|
||||
}
|
||||
|
||||
idxRows, err := db.Query("SELECT name, \"unique\", partial FROM pragma_index_list('student')")
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
defer idxRows.Close()
|
||||
var idx []string
|
||||
for idxRows.Next() {
|
||||
var name string
|
||||
var uniq, partial int
|
||||
if err := idxRows.Scan(&name, &uniq, &partial); err != nil {
|
||||
return "", err
|
||||
}
|
||||
idx = append(idx, fmt.Sprintf("%s:%d:%d", name, uniq, partial))
|
||||
}
|
||||
if err := idxRows.Err(); err != nil {
|
||||
return "", err
|
||||
}
|
||||
// Index order is not guaranteed by SQLite; sort so it cannot cause a false
|
||||
// difference.
|
||||
sort.Strings(idx)
|
||||
|
||||
return strings.Join(cols, "|") + "\n" + strings.Join(idx, "|"), nil
|
||||
}
|
||||
|
||||
// Field and record separators. The values are hashed with explicit delimiters
|
||||
// so that ("ab","c") and ("a","bc") cannot produce the same digest — joining
|
||||
// them bare would make a shifted field boundary invisible.
|
||||
const (
|
||||
fieldSep = "\x1f"
|
||||
recordSep = "\x1e"
|
||||
)
|
||||
|
||||
// ser renders one value canonically.
|
||||
//
|
||||
// NULL gets a sentinel no real value can collide with, and a whole-numbered
|
||||
// REAL is rendered with one decimal place so that a column holding 3 and one
|
||||
// holding 3.0 compare equal, as SQLite considers them.
|
||||
func ser(v any) string {
|
||||
switch t := v.(type) {
|
||||
case nil:
|
||||
return "\x00NULL"
|
||||
case int64:
|
||||
return strconv.FormatInt(t, 10)
|
||||
case float64:
|
||||
if t == math.Trunc(t) && !math.IsInf(t, 0) {
|
||||
return strconv.FormatFloat(t, 'f', 1, 64)
|
||||
}
|
||||
return strconv.FormatFloat(t, 'g', -1, 64)
|
||||
case []byte:
|
||||
return string(t)
|
||||
case string:
|
||||
return t
|
||||
default:
|
||||
return fmt.Sprint(t)
|
||||
}
|
||||
}
|
||||
|
||||
// tableHash is a rolling SHA-256 over every row, ordered by primary key.
|
||||
//
|
||||
// Streamed rather than materialised: 877k rows by 22 columns would otherwise be
|
||||
// a large amount of memory for no benefit.
|
||||
func tableHash(db *sql.DB, cols []string) (string, error) {
|
||||
rows, err := db.Query("SELECT * FROM student ORDER BY so_bao_danh")
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
defer rows.Close()
|
||||
|
||||
h := sha256.New()
|
||||
vals := make([]any, len(cols))
|
||||
ptrs := make([]any, len(cols))
|
||||
for i := range vals {
|
||||
ptrs[i] = &vals[i]
|
||||
}
|
||||
|
||||
for rows.Next() {
|
||||
if err := rows.Scan(ptrs...); err != nil {
|
||||
return "", err
|
||||
}
|
||||
for i := range vals {
|
||||
io.WriteString(h, ser(vals[i]))
|
||||
io.WriteString(h, fieldSep)
|
||||
}
|
||||
io.WriteString(h, recordSep)
|
||||
}
|
||||
if err := rows.Err(); err != nil {
|
||||
return "", err
|
||||
}
|
||||
return hex.EncodeToString(h.Sum(nil)), nil
|
||||
}
|
||||
|
||||
// firstDifferences walks both tables in step and reports where they diverge,
|
||||
// so a failure names a row and a column rather than only a digest.
|
||||
func firstDifferences(a, b *sql.DB, cols []string, limit int) ([]string, error) {
|
||||
ra, err := a.Query("SELECT * FROM student ORDER BY so_bao_danh")
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer ra.Close()
|
||||
rb, err := b.Query("SELECT * FROM student ORDER BY so_bao_danh")
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer rb.Close()
|
||||
|
||||
scan := func(rows *sql.Rows) ([]string, bool, error) {
|
||||
if !rows.Next() {
|
||||
return nil, false, rows.Err()
|
||||
}
|
||||
vals := make([]any, len(cols))
|
||||
ptrs := make([]any, len(cols))
|
||||
for i := range vals {
|
||||
ptrs[i] = &vals[i]
|
||||
}
|
||||
if err := rows.Scan(ptrs...); err != nil {
|
||||
return nil, false, err
|
||||
}
|
||||
out := make([]string, len(cols))
|
||||
for i := range vals {
|
||||
out[i] = ser(vals[i])
|
||||
}
|
||||
return out, true, nil
|
||||
}
|
||||
|
||||
var out []string
|
||||
for len(out) < limit {
|
||||
va, okA, err := scan(ra)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
vb, okB, err := scan(rb)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if !okA || !okB {
|
||||
break
|
||||
}
|
||||
for i, c := range cols {
|
||||
if va[i] != vb[i] {
|
||||
out = append(out, fmt.Sprintf(" so_bao_danh=%s %s: a=%q b=%q", va[0], c, va[i], vb[i]))
|
||||
break
|
||||
}
|
||||
}
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
@@ -0,0 +1,165 @@
|
||||
package verify
|
||||
|
||||
import (
|
||||
"database/sql"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"github.com/tiennm99/thptqg/assembler/internal/registry"
|
||||
)
|
||||
|
||||
// makeDB writes a miniature student table so the comparison logic can be
|
||||
// exercised without building a real 877k-row database.
|
||||
func makeDB(t *testing.T, dir, id string, rows [][3]any) {
|
||||
t.Helper()
|
||||
path := filepath.Join(dir, id+".db")
|
||||
if err := os.MkdirAll(dir, 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
db, err := sql.Open(driverName, path)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
defer db.Close()
|
||||
|
||||
if _, err := db.Exec(`CREATE TABLE student (
|
||||
so_bao_danh TEXT PRIMARY KEY,
|
||||
ho_ten TEXT NOT NULL,
|
||||
toan REAL
|
||||
); CREATE INDEX idx_ho_ten ON student(ho_ten);`); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
for _, r := range rows {
|
||||
if _, err := db.Exec("INSERT INTO student VALUES (?,?,?)", r[0], r[1], r[2]); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
var base = [][3]any{
|
||||
{"001", "Nguyễn Văn A", 8.0},
|
||||
{"002", "Trần Thị B", 7.25},
|
||||
{"003", "Lê Văn C", nil},
|
||||
}
|
||||
|
||||
func compare(t *testing.T, a, b [][3]any) Result {
|
||||
t.Helper()
|
||||
dirA, dirB := t.TempDir(), t.TempDir()
|
||||
makeDB(t, dirA, "x", a)
|
||||
makeDB(t, dirB, "x", b)
|
||||
|
||||
res, err := Compare([]registry.Dataset{{ID: "x"}}, dirA, dirB)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return res[0]
|
||||
}
|
||||
|
||||
func TestIdenticalDatabasesMatch(t *testing.T) {
|
||||
got := compare(t, base, base)
|
||||
if !got.OK() {
|
||||
t.Fatalf("identical databases reported problems: %v", got.Problems)
|
||||
}
|
||||
if got.HashA != got.HashB || got.HashA == "" {
|
||||
t.Errorf("hashes should be equal and non-empty: %q %q", got.HashA, got.HashB)
|
||||
}
|
||||
if got.RowsA != 3 {
|
||||
t.Errorf("RowsA = %d, want 3", got.RowsA)
|
||||
}
|
||||
}
|
||||
|
||||
// TestOneChangedScoreIsCaught is the whole point: a single altered field, with
|
||||
// the row count and every non-NULL count unchanged, must still fail.
|
||||
func TestOneChangedScoreIsCaught(t *testing.T) {
|
||||
changed := [][3]any{{"001", "Nguyễn Văn A", 8.25}, base[1], base[2]}
|
||||
got := compare(t, base, changed)
|
||||
|
||||
if got.OK() {
|
||||
t.Fatal("a changed score was not detected")
|
||||
}
|
||||
if got.RowsA != got.RowsB {
|
||||
t.Error("row counts should still match — that is what makes this hard")
|
||||
}
|
||||
joined := strings.Join(got.Problems, "; ")
|
||||
if !strings.Contains(joined, "full-table hash differs") {
|
||||
t.Errorf("problems = %v", got.Problems)
|
||||
}
|
||||
// The diagnosis must name the row and the column.
|
||||
diff := strings.Join(got.FirstDiff, "\n")
|
||||
if !strings.Contains(diff, "so_bao_danh=001") || !strings.Contains(diff, "toan") {
|
||||
t.Errorf("FirstDiff should locate the change, got: %v", got.FirstDiff)
|
||||
}
|
||||
}
|
||||
|
||||
// TestNullVersusValueIsCaught: a value becoming NULL changes the per-column
|
||||
// count as well, so both signals should fire.
|
||||
func TestNullVersusValueIsCaught(t *testing.T) {
|
||||
nulled := [][3]any{base[0], base[1], {"003", "Lê Văn C", 5.0}}
|
||||
got := compare(t, base, nulled)
|
||||
if got.OK() {
|
||||
t.Fatal("a NULL becoming a value was not detected")
|
||||
}
|
||||
if !strings.Contains(strings.Join(got.Problems, ";"), "non-NULL count for toan") {
|
||||
t.Errorf("expected a per-column count difference, got %v", got.Problems)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRowCountDifferenceIsCaught(t *testing.T) {
|
||||
got := compare(t, base, base[:2])
|
||||
if got.OK() {
|
||||
t.Fatal("a missing row was not detected")
|
||||
}
|
||||
if !strings.Contains(strings.Join(got.Problems, ";"), "row count 3 vs 2") {
|
||||
t.Errorf("problems = %v", got.Problems)
|
||||
}
|
||||
}
|
||||
|
||||
// TestTextShiftAcrossFieldsIsCaught guards the field separator. Hashing the
|
||||
// values joined bare would make ("ab","c") and ("a","bc") identical, so a value
|
||||
// shifted across a column boundary would slip through.
|
||||
func TestTextShiftAcrossFieldsIsCaught(t *testing.T) {
|
||||
a := [][3]any{{"001", "ab", 1.0}}
|
||||
b := [][3]any{{"001", "a", 1.0}}
|
||||
// Same idea at the boundary between so_bao_danh and ho_ten.
|
||||
c := [][3]any{{"01", "23", 1.0}}
|
||||
d := [][3]any{{"012", "3", 1.0}}
|
||||
|
||||
if compare(t, a, b).OK() {
|
||||
t.Error("a shortened text value was not detected")
|
||||
}
|
||||
if compare(t, c, d).OK() {
|
||||
t.Error("a value shifted across a field boundary was not detected")
|
||||
}
|
||||
}
|
||||
|
||||
func TestMissingDatabaseIsAnError(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
makeDB(t, dir, "x", base)
|
||||
if _, err := Compare([]registry.Dataset{{ID: "x"}}, dir, t.TempDir()); err == nil {
|
||||
t.Fatal("comparing against a directory with no database must fail")
|
||||
}
|
||||
}
|
||||
|
||||
func TestSer(t *testing.T) {
|
||||
for _, tc := range []struct {
|
||||
in any
|
||||
want string
|
||||
}{
|
||||
{nil, "\x00NULL"},
|
||||
{int64(7), "7"},
|
||||
{8.0, "8.0"}, // whole REALs get a fixed rendering
|
||||
{7.25, "7.25"},
|
||||
{"x", "x"},
|
||||
{[]byte("y"), "y"},
|
||||
} {
|
||||
if got := ser(tc.in); got != tc.want {
|
||||
t.Errorf("ser(%v) = %q, want %q", tc.in, got, tc.want)
|
||||
}
|
||||
}
|
||||
// A NULL must not collide with any real text value.
|
||||
if ser(nil) == ser("NULL") || ser(nil) == ser("") {
|
||||
t.Error("the NULL sentinel collides with a real value")
|
||||
}
|
||||
}
|
||||
@@ -52,7 +52,7 @@ func resolveFixture(t *testing.T, src Source) []File {
|
||||
// parser sorts its inputs and inserts last-wins, so filenames decide which
|
||||
// row survives a duplicate exam number. If extraction or naming drifted, a
|
||||
// re-crawl could rebuild a database with the same row count and different
|
||||
// content, which the row-count guard in build-db.js would not catch.
|
||||
// content, which the assembler's row-count guard would not catch.
|
||||
//
|
||||
// The committed data/<id>/ is the oracle: reading the source article and
|
||||
// applying the source's naming rule must reproduce it exactly, both directions.
|
||||
|
||||
+40
-17
@@ -133,10 +133,22 @@ the 2017 configs listed 14. Neither list was complete: candidates could sit
|
||||
German, Japanese and Russian in both years, so 1,691 students ended up with **no
|
||||
foreign-language score at all**.
|
||||
|
||||
Running all 16 patterns everywhere recovered them — 182 Russian in 2016, and
|
||||
German and Japanese across all three 2017 generations. See
|
||||
`plans/reports/parser-parity-result.md` for the evidence that these are real
|
||||
scores rather than false matches.
|
||||
Running all 16 patterns everywhere recovered them: 182 × `tieng_nga` in 2016,
|
||||
and 93 × `tieng_duc` plus 512 × `tieng_nhat` in 2017.
|
||||
|
||||
Four things established that these are real scores and not false regex matches,
|
||||
checked when the change was made:
|
||||
|
||||
- Every student holds zero or exactly one foreign language, never two, so
|
||||
nothing was double-counted.
|
||||
- Each affected student had **all** language columns NULL beforehand — e.g. SBD
|
||||
`01003198` went from all-NULL to `tieng_duc = 8`.
|
||||
- The counts tracked dataset size across the three 2017 publications that then
|
||||
existed (93/85/22 German, 512/484/313 Japanese).
|
||||
- Every recovered value is an ordinary exam score in the 0–10 range.
|
||||
|
||||
Row counts and the non-NULL counts of all 18 pre-existing columns were unchanged
|
||||
across every dataset; only the four language columns gained values.
|
||||
|
||||
## Per-dataset quirks
|
||||
|
||||
@@ -166,19 +178,26 @@ figure in the table above, and each `.db.gz` must be at least 90% of its usual
|
||||
size, or the build fails rather than publishing. That guard is the reason a
|
||||
truncated dataset cannot reach the site with a green pipeline.
|
||||
|
||||
For a deeper check, `parser/scripts/differential-parity.mjs` compares two sets
|
||||
of databases field-by-field — row counts, per-column non-NULL counts, a
|
||||
full-table SHA-256 over every row ordered by `so_bao_danh`, schema metadata, and
|
||||
build stdout:
|
||||
For a deeper check, `assemble verify` compares two sets of built databases
|
||||
field by field — row counts, per-column non-NULL counts, schema metadata, and a
|
||||
full-table SHA-256 over every row ordered by `so_bao_danh`:
|
||||
|
||||
```bash
|
||||
node parser/scripts/differential-parity.mjs \
|
||||
--rust /path/to/a-{id}.db --go /path/to/b-{id}.db
|
||||
cp -r .build/public/db /tmp/before # keep the databases you have
|
||||
go -C assembler run ./cmd/assemble db # rebuild
|
||||
go -C assembler run ./cmd/assemble verify /tmp/before .build/public/db
|
||||
```
|
||||
|
||||
It exits non-zero on any mismatch and fails loudly if a dataset is missing rather
|
||||
than skipping it. Written for the Rust-to-Go migration, it works for any two
|
||||
builds. Uses the built-in `node:sqlite`, so it needs no dependencies.
|
||||
Each side is a directory of `<id>.db.gz` (or `<id>.db`); compressed databases are
|
||||
expanded to a temporary file automatically. It exits non-zero on any mismatch,
|
||||
names the first differing rows and columns, and fails rather than skipping when a
|
||||
dataset is absent from either side — silently comparing one of two datasets is
|
||||
how a gate passes without proving anything.
|
||||
|
||||
This is the only check on database *content*. The reader oracle below covers
|
||||
reading the spreadsheets and the row-count guard covers how many rows came out,
|
||||
but a change in transform or writer logic can alter what is in those rows while
|
||||
both of those still pass.
|
||||
|
||||
`parser/internal/reader` additionally carries a frozen oracle of per-file
|
||||
cell-dump hashes covering all 182 inputs; `go -C parser test ./...` fails if any single
|
||||
@@ -194,8 +213,8 @@ go -C assembler run ./cmd/assemble db 2017
|
||||
|
||||
The row-count guard in `build:db` confirms the rebuild matches the expected
|
||||
total. That guard checks the count only, so if the crawl was expected to change
|
||||
the data, compare content with `differential-parity.mjs` against a copy of the
|
||||
previous database rather than trusting the count.
|
||||
the data, compare content with `assemble verify` against a copy of the previous
|
||||
database rather than trusting the count.
|
||||
|
||||
## Removed scripts
|
||||
|
||||
@@ -203,8 +222,12 @@ previous database rather than trusting the count.
|
||||
were dropped with the Rust parser. The first two had been broken since before the
|
||||
repo was unified (a hardcoded Windows path in one, an undeclared
|
||||
`better-sqlite3` dependency in the other) and neither had any automated caller.
|
||||
The latter two are superseded by `differential-parity.mjs`, which compares more
|
||||
and cannot silently skip a dataset.
|
||||
The latter two are superseded by `assemble verify`, which compares more and
|
||||
cannot silently skip a dataset. `differential-parity.mjs`, which was that
|
||||
comparator, has itself been folded into the assembler as
|
||||
`assembler/internal/verify` — the pipeline is Go outside `web/`, and the port
|
||||
also fixed a weakness: the JavaScript version hashed each row's fields joined
|
||||
bare, so a value shifted across a column boundary produced the same digest.
|
||||
|
||||
`crawl-baotintuc.js` was not dropped but rewritten as the Go `crawler/` module,
|
||||
producing the same local filenames. Two changes beyond the port:
|
||||
|
||||
@@ -8,7 +8,7 @@ One-time setup: **Settings → Pages → Source: GitHub Actions**.
|
||||
## What the workflow does
|
||||
|
||||
1. Checkout, Go toolchain, Node 24, `npm ci` in `web/`
|
||||
2. Parser and crawler test suites, web lint, `govulncheck` over all three modules
|
||||
2. Parser, crawler and assembler test suites, web lint, `govulncheck` over all three modules
|
||||
3. `go -C assembler run ./cmd/assemble` — the whole pipeline: compile the
|
||||
parser, build and verify each database, compress it into `.build/public/db/`,
|
||||
run the Vite build, assemble `_site/`
|
||||
@@ -85,7 +85,7 @@ artifact — one missing line away from publishing it.
|
||||
- **No server-side compression is assumed.** The app fetches the `.gz` bytes
|
||||
directly rather than relying on `Content-Encoding: gzip`; Pages does not
|
||||
reliably compress arbitrary paths on the fly.
|
||||
- **Total artifact is about 177 MB**, well inside the 1 GB site limit.
|
||||
- **Total artifact is about 93 MB**, well inside the 1 GB site limit.
|
||||
|
||||
## Rollback
|
||||
|
||||
|
||||
+9
-2
@@ -49,7 +49,7 @@ containing it.
|
||||
The port was gated on a field-by-field comparison of both implementations across
|
||||
the four datasets that existed then — 3,265,641 rows with identical full-table
|
||||
SHA-256, identical per-column non-NULL counts, identical schema metadata and
|
||||
identical build stdout. `scripts/differential-parity.mjs` is that comparator and
|
||||
identical build stdout. `assemble verify` is that comparator today and
|
||||
still runs against any two sets of databases.
|
||||
|
||||
Behaviour was matched bug-for-bug, deliberately. Several quirks look like
|
||||
@@ -70,7 +70,14 @@ Each has a test naming it, so none can be tidied away by accident.
|
||||
`testdata/reader-fidelity-hashes.tsv` holds a SHA-256 per input file over a
|
||||
canonical dump of every cell of every sheet. It is **frozen**: it was produced
|
||||
by the Rust reader, which no longer exists, so it cannot be regenerated. It
|
||||
still fails if any single cell of any of the 182 files reads differently.
|
||||
still fails if any single cell of any input file reads differently.
|
||||
|
||||
A mismatch names the file but not the cell. `cmd/dumpcells` prints the stream
|
||||
the hash is taken over, so two runs can be diffed:
|
||||
|
||||
```bash
|
||||
go -C parser run ./cmd/dumpcells ../data/2017/an-giang.xls out.tsv
|
||||
```
|
||||
|
||||
The assembler refuses to publish a database whose row count does not match
|
||||
the known figure, or whose artifact is under 90% of its usual size.
|
||||
@@ -1,6 +1,11 @@
|
||||
// Command dumpcells emits the canonical cell rendering of a spreadsheet, for
|
||||
// comparison against the Rust/calamine ground truth produced by
|
||||
// parser/examples/dump_cells.rs.
|
||||
// Command dumpcells emits the canonical cell rendering of a spreadsheet.
|
||||
//
|
||||
// It exists to diagnose a reader-fidelity failure. That suite compares a
|
||||
// SHA-256 per input file against a frozen oracle, so a mismatch says which file
|
||||
// changed but not which cell; this prints the stream the hash is taken over, so
|
||||
// two runs can be diffed. It was originally written to compare against the
|
||||
// Rust/calamine ground truth from parser/examples/dump_cells.rs, which was
|
||||
// removed with the Rust tree — the hashes it produced are what remains.
|
||||
//
|
||||
// The canonical stream carries geometry and rendered cell values only. The
|
||||
// calamine Data variant is deliberately excluded: Data::Empty and
|
||||
|
||||
@@ -1,8 +1,8 @@
|
||||
// Command xlsxread reads .xls/.xlsx files and builds SQLite databases for the
|
||||
// thptqg datasets.
|
||||
//
|
||||
// The CLI contract is fixed by parser/scripts/build-db.js and must match the
|
||||
// Rust binary exactly:
|
||||
// The CLI contract is fixed by the assembler, which drives this binary, and it
|
||||
// matched the Rust binary it replaced exactly:
|
||||
//
|
||||
// xlsxread build --schema <config.yml> --input <dir> --output <db>
|
||||
// xlsxread audit --schema <config.yml> --input <dir> --db <db>
|
||||
|
||||
@@ -2,8 +2,8 @@
|
||||
//
|
||||
// The config deliberately carries no SQL. The table shape, the INSERT and the
|
||||
// subject regexes are identical for every dataset and live in internal/schema;
|
||||
// keeping them here meant four copies of the same DDL, which is how the 2016 and
|
||||
// 2017 schemas drifted apart.
|
||||
// keeping them here meant one copy of the DDL per dataset, which is how the
|
||||
// 2016 and 2017 schemas drifted apart.
|
||||
package config
|
||||
|
||||
import (
|
||||
|
||||
@@ -309,7 +309,7 @@ func Detect2016(cfg *config.DatasetConfig, inputDir, outputPath string) error {
|
||||
if err := tx.Commit(); err != nil {
|
||||
return fmt.Errorf("commit: %w", err)
|
||||
}
|
||||
return writer.Finish(db, outputPath, st, label, false)
|
||||
return writer.Finish(db, outputPath, st)
|
||||
}
|
||||
|
||||
// processFile2016 detects the layout per SHEET and processes that sheet's rows.
|
||||
|
||||
@@ -147,7 +147,6 @@ func Standard(cfg *config.DatasetConfig, inputDir, outputPath string) error {
|
||||
}
|
||||
defer db.Close()
|
||||
|
||||
isOld2 := strings.Contains(label, "old2")
|
||||
stripBlank := cfg.Reader.StripBlankRows
|
||||
|
||||
// One transaction spans the whole dataset directory (main.rs:120,184).
|
||||
@@ -230,7 +229,7 @@ func Standard(cfg *config.DatasetConfig, inputDir, outputPath string) error {
|
||||
}
|
||||
|
||||
// VACUUM only after COMMIT — SQLite refuses it inside a transaction.
|
||||
return writer.Finish(db, outputPath, st, label, isOld2)
|
||||
return writer.Finish(db, outputPath, st)
|
||||
}
|
||||
|
||||
// cellAt returns the trimmed cell at idx, or "" when out of range — the
|
||||
|
||||
@@ -30,7 +30,7 @@ import (
|
||||
// under -short, which is how to iterate without paying ~77s to re-read 418 MB.
|
||||
func TestReaderFidelity(t *testing.T) {
|
||||
if testing.Short() {
|
||||
t.Skip("-short: skipping the 299-file corpus sweep")
|
||||
t.Skip("-short: skipping the full-corpus sweep")
|
||||
}
|
||||
root := repoRoot(t)
|
||||
manifest := filepath.Join(root, "parser", "testdata", "reader-fidelity-hashes.tsv")
|
||||
|
||||
@@ -2,9 +2,12 @@
|
||||
// contract, mirroring parser/src/reader.rs (which wraps calamine's
|
||||
// open_workbook_auto for both formats).
|
||||
//
|
||||
// The contract is deliberately row-streaming rather than whole-workbook
|
||||
// materialising: data/2017/ha-noi.xls alone holds 72k rows across two sheets,
|
||||
// and the Rust original keeps at most one sheet range live at a time.
|
||||
// The *contract* is row-at-a-time: callers receive one row per callback and
|
||||
// never hold the workbook. The two implementations do not honour that in
|
||||
// memory — both decode the whole workbook into [][][]Cell up front, because
|
||||
// neither underlying library exposes a usable row cursor. data/2017/ha-noi.xls
|
||||
// alone holds 72k rows across two sheets, so a caller that assumed streaming
|
||||
// memory here would be wrong.
|
||||
package reader
|
||||
|
||||
import (
|
||||
@@ -27,7 +30,7 @@ type Cell struct {
|
||||
|
||||
// Sheet carries a sheet's identity and geometry. Height and Width describe the
|
||||
// used range, matching calamine's Range::height()/width(); every used range in
|
||||
// the corpus starts at (0,0), verified across all 299 files.
|
||||
// the corpus starts at (0,0), verified across every input file.
|
||||
type Sheet struct {
|
||||
Index int
|
||||
Name string
|
||||
|
||||
@@ -1,8 +1,7 @@
|
||||
// Package schema is the single source of truth for the SQL shape of every
|
||||
// dataset — a direct port of parser/src/schema.rs.
|
||||
//
|
||||
// All four datasets (2016, 2017, 2017-old, 2017-old2) write into the same
|
||||
// 22-column student table. Columns a dataset has no data for bind NULL.
|
||||
// Every dataset writes into the same 22-column student table. Columns a dataset has no data for bind NULL.
|
||||
//
|
||||
// Column provenance:
|
||||
//
|
||||
|
||||
@@ -117,12 +117,12 @@ type Stats struct {
|
||||
//
|
||||
// VACUUM must run AFTER the transaction commits — SQLite refuses it inside one.
|
||||
//
|
||||
// The wording branches on datasetLabel, which Rust derives from the input
|
||||
// directory's basename (main.rs:98-101). That makes the output depend on a
|
||||
// filesystem path rather than on config; it is reproduced here for parity, and
|
||||
// the caller passes the label explicitly so tests are not at the mercy of a
|
||||
// temp-directory name.
|
||||
func Finish(db *sql.DB, dbPath string, st Stats, datasetLabel string, isOld2 bool) error {
|
||||
// The Rust original varied the wording by dataset, keying off the input
|
||||
// directory's basename (main.rs:98-101) — one phrasing for the 2017-old2
|
||||
// export, another for anything with "old" in its name. Both of those datasets
|
||||
// have been removed, so only one branch was ever taken; the wording below is
|
||||
// exactly what 2016 and 2017 have always printed.
|
||||
func Finish(db *sql.DB, dbPath string, st Stats) error {
|
||||
if _, err := db.Exec("VACUUM"); err != nil {
|
||||
return fmt.Errorf("vacuum: %w", err)
|
||||
}
|
||||
@@ -135,22 +135,13 @@ func Finish(db *sql.DB, dbPath string, st Stats, datasetLabel string, isOld2 boo
|
||||
insertable := st.SourceRows - st.Skipped
|
||||
|
||||
fmt.Println()
|
||||
if isOld2 {
|
||||
fmt.Printf("Source non-blank data rows: %d\n", st.SourceRows)
|
||||
fmt.Printf(" skipped (empty/non-numeric SBD): %d\n", st.Skipped)
|
||||
} else {
|
||||
fmt.Printf("Source data rows (post-header): %d\n", st.SourceRows)
|
||||
if containsOld(datasetLabel) {
|
||||
fmt.Printf(" skipped (empty/non-numeric SBD): %d\n", st.Skipped)
|
||||
} else {
|
||||
fmt.Printf(" skipped (empty/invalid): %d\n", st.Skipped)
|
||||
}
|
||||
}
|
||||
fmt.Printf(" insertable: %d\n", insertable)
|
||||
fmt.Printf(" insert errors: %d\n", st.Errors)
|
||||
fmt.Printf("DB rows (distinct SBD): %d\n", dbCount)
|
||||
|
||||
if !containsOld(datasetLabel) && st.Errors == 0 {
|
||||
if st.Errors == 0 {
|
||||
gap := int64(insertable) - dbCount
|
||||
if gap == 0 {
|
||||
fmt.Println("Audit: OK — every source row made it in.")
|
||||
@@ -166,12 +157,3 @@ func Finish(db *sql.DB, dbPath string, st Stats, datasetLabel string, isOld2 boo
|
||||
fmt.Printf("Size: %.1f MB\n", float64(size)/1024.0/1024.0)
|
||||
return nil
|
||||
}
|
||||
|
||||
func containsOld(label string) bool {
|
||||
for i := 0; i+3 <= len(label); i++ {
|
||||
if label[i:i+3] == "old" {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
Binary file not shown.
+1
-1
@@ -5,7 +5,7 @@
|
||||
# FROZEN. These hashes were produced by the Rust/calamine parser, which has
|
||||
# since been removed, so they cannot be regenerated. They remain a real
|
||||
# regression guard: any change to the Go reader that alters a single cell of any
|
||||
# of the 299 input files fails the suite. If the inputs themselves ever change,
|
||||
# input file listed below fails the suite. If the inputs themselves ever change,
|
||||
# the affected rows must be re-derived deliberately and reviewed, not refreshed
|
||||
# in bulk.
|
||||
data/2016/023718c7d3cf7ace3a7116fabb12bd9cdhyduoccantho-1468920829104.xlsx fc3afd8da9ed28b0fc177192fab7b19a06535cb45d86841835d8e8828fc61c1c
|
||||
|
||||
|
File diff suppressed because it is too large.
Load diff
File diff suppressed because it is too large.
Load diff
@@ -1,85 +0,0 @@
|
||||
# Parser parity result
|
||||
|
||||
**Archived record.** This documents a verification run made on 2026-08-13,
|
||||
against a tree that had a Rust parser, npm-script entry points and four
|
||||
datasets — none of which exist now. It is kept because `docs/data-pipeline.md`
|
||||
cites it as the evidence that the recovered foreign-language scores are real
|
||||
rather than false regex matches. Do not expect the commands below to run.
|
||||
|
||||
Comparison of the databases the unified pipeline ships against a baseline built
|
||||
from the pre-refactor code. Run on 2026-08-13 against the tree as it then stood
|
||||
(`npm run build:rust && npm run build:db`), gate exit code 0.
|
||||
|
||||
Inputs:
|
||||
|
||||
- `plans/reports/parser-parity-baseline.json` — built with the two separate
|
||||
crates, before any change
|
||||
- `plans/reports/parser-parity-current.json` — decompressed from
|
||||
`.build/public/db/*.db.gz`, i.e. the exact bytes published
|
||||
|
||||
## Result
|
||||
|
||||
| Dataset | Rows | Columns | Size |
|
||||
| --- | --- | --- | --- |
|
||||
| 2016 | 877,461 | 18 → 22 | +1.4% |
|
||||
| 2017 | 861,068 | 18 → 22 | +2.2% |
|
||||
| 2017-old | 847,348 | 18 → 22 | +2.2% |
|
||||
| 2017-old2 | 679,764 | 18 → 22 | +2.2% |
|
||||
|
||||
Unchanged, for every dataset:
|
||||
|
||||
- row count
|
||||
- non-NULL count for all 18 pre-existing columns
|
||||
- every field of a deterministic student sample (SBDs ending `0000`)
|
||||
|
||||
## Approved recoveries
|
||||
|
||||
Unifying the subject regexes recovered 1,691 foreign-language scores that the
|
||||
old per-year configs discarded. The 2016 config listed 12 subject patterns and
|
||||
the 2017 configs 14; neither list was complete, so candidates who sat German,
|
||||
Japanese or Russian ended up with no foreign-language score at all.
|
||||
|
||||
| Dataset | Recovered |
|
||||
| --- | --- |
|
||||
| 2016 | 182 × `tieng_nga` |
|
||||
| 2017 | 93 × `tieng_duc`, 512 × `tieng_nhat` |
|
||||
| 2017-old | 85 × `tieng_duc`, 484 × `tieng_nhat` |
|
||||
| 2017-old2 | 22 × `tieng_duc`, 313 × `tieng_nhat` |
|
||||
|
||||
Confirmed real rather than spurious matches:
|
||||
|
||||
- Every student in all four datasets holds zero or exactly one foreign
|
||||
language — never two — so nothing is duplicated.
|
||||
- Each affected student had all language columns NULL beforehand. SBD
|
||||
`01003198` went from all-NULL to `tieng_duc = 8`.
|
||||
- Counts track dataset size across the three 2017 generations (93/85/22
|
||||
German, 512/484/313 Japanese).
|
||||
- Values are ordinary exam scores in the 0–10 range.
|
||||
|
||||
These exact counts are pinned as `APPROVED_RECOVERY` in
|
||||
`parser/scripts/verify-parity.js`. Any other newly-populated column, or drift
|
||||
in these numbers, still fails the gate.
|
||||
|
||||
## Columns that must stay empty
|
||||
|
||||
Verified 0 non-NULL:
|
||||
|
||||
- 2016: `khtn`, `khxh`, `gdcd`
|
||||
- 2017, 2017-old, 2017-old2: `ten_cum_thi`, `gioi_tinh`
|
||||
|
||||
## Reproducing
|
||||
|
||||
```sh
|
||||
npm run build:rust && npm run build:db
|
||||
node parser/scripts/db-stats.js \
|
||||
2016=<db> 2017=<db> 2017-old=<db> 2017-old2=<db> > current.json
|
||||
node parser/scripts/verify-parity.js \
|
||||
plans/reports/parser-parity-baseline.json current.json
|
||||
```
|
||||
|
||||
The baseline cannot be regenerated — it required the two pre-refactor crates,
|
||||
which no longer exist. It is committed for that reason.
|
||||
|
||||
## Unresolved questions
|
||||
|
||||
None.
|
||||
+6
-2
@@ -2,7 +2,6 @@
|
||||
<html lang="vi">
|
||||
<head>
|
||||
<meta charset="UTF-8" />
|
||||
<link rel="icon" type="image/svg+xml" href="/favicon.svg" />
|
||||
<meta name="viewport" content="width=device-width, initial-scale=1.0" />
|
||||
<link rel="preconnect" href="https://fonts.googleapis.com" />
|
||||
<link rel="preconnect" href="https://fonts.gstatic.com" crossorigin />
|
||||
@@ -10,7 +9,12 @@
|
||||
rel="stylesheet"
|
||||
href="https://fonts.googleapis.com/css2?family=Be+Vietnam+Pro:wght@400;500;600;700&display=swap"
|
||||
/>
|
||||
<title>Tra cứu điểm thi THPT QG 2017</title>
|
||||
<!--
|
||||
One index.html is copied to every route, so this title has to be the one
|
||||
that suits them all; the app replaces it with the dataset's own title once
|
||||
it knows which route it is on.
|
||||
-->
|
||||
<title>Tra cứu điểm thi THPT Quốc gia</title>
|
||||
</head>
|
||||
<body>
|
||||
<div id="root"></div>
|
||||
|
||||
@@ -14,7 +14,6 @@
|
||||
--focus-ring: 0 0 0 3px rgba(59, 79, 228, 0.35);
|
||||
--radius: 10px;
|
||||
--radius-sm: 6px;
|
||||
--shadow-sm: 0 1px 2px rgba(15, 20, 40, 0.05);
|
||||
--shadow-md: 0 6px 20px rgba(15, 20, 40, 0.08);
|
||||
|
||||
/* Score tier palette — TFT rarity ladder (white → green → blue → purple → gold → prismatic).
|
||||
@@ -49,7 +48,6 @@
|
||||
--primary-hover: #879aff;
|
||||
--primary-ink: #0f1220;
|
||||
--focus-ring: 0 0 0 3px rgba(109, 130, 255, 0.45);
|
||||
--shadow-sm: 0 1px 2px rgba(0, 0, 0, 0.4);
|
||||
--shadow-md: 0 6px 20px rgba(0, 0, 0, 0.55);
|
||||
|
||||
--tier-common-bg: #242838; --tier-common-ink: #c5cad8; --tier-common-accent: #4a5163;
|
||||
@@ -853,16 +851,7 @@ footer {
|
||||
font-size: 0.9rem;
|
||||
}
|
||||
|
||||
.hub-archive h2 {
|
||||
font-size: 0.95rem;
|
||||
font-weight: 600;
|
||||
color: var(--text-muted);
|
||||
margin-bottom: 0;
|
||||
}
|
||||
|
||||
.hub-list-compact {
|
||||
margin-top: 0.75rem;
|
||||
}
|
||||
|
||||
.hub-back {
|
||||
margin: 0.35rem 0 0;
|
||||
|
||||
@@ -147,6 +147,12 @@ function DatasetApp({ dataset }) {
|
||||
writeUrlQuery("");
|
||||
}, []);
|
||||
|
||||
// The assembler copies one index.html to every route, so its static <title>
|
||||
// cannot name a dataset. Set the real one once the route is resolved.
|
||||
useEffect(() => {
|
||||
document.title = dataset.title;
|
||||
}, [dataset.title]);
|
||||
|
||||
return (
|
||||
<div className="app">
|
||||
<header>
|
||||
|
||||
@@ -1 +0,0 @@
|
||||
<svg xmlns="http://www.w3.org/2000/svg" width="77" height="47" fill="none" aria-labelledby="vite-logo-title" viewBox="0 0 77 47"><title id="vite-logo-title">Vite</title><style>.parenthesis{fill:#000}@media (prefers-color-scheme:dark){.parenthesis{fill:#fff}}</style><path fill="#9135ff" d="M40.151 45.71c-.663.844-2.02.374-2.02-.699V34.708a2.26 2.26 0 0 0-2.262-2.262H24.493c-.92 0-1.457-1.04-.92-1.788l7.479-10.471c1.07-1.498 0-3.578-1.842-3.578H15.443c-.92 0-1.456-1.04-.92-1.788l9.696-13.576c.213-.297.556-.474.92-.474h28.894c.92 0 1.456 1.04.92 1.788l-7.48 10.472c-1.07 1.497 0 3.578 1.842 3.578h11.376c.944 0 1.474 1.087.89 1.83L40.153 45.712z"/><mask id="a" width="48" height="47" x="14" y="0" maskUnits="userSpaceOnUse" style="mask-type:alpha"><path fill="#000" d="M40.047 45.71c-.663.843-2.02.374-2.02-.699V34.708a2.26 2.26 0 0 0-2.262-2.262H24.389c-.92 0-1.457-1.04-.92-1.788l7.479-10.472c1.07-1.497 0-3.578-1.842-3.578H15.34c-.92 0-1.456-1.04-.92-1.788l9.696-13.575c.213-.297.556-.474.92-.474H53.93c.92 0 1.456 1.04.92 1.788L47.37 13.03c-1.07 1.498 0 3.578 1.842 3.578h11.376c.944 0 1.474 1.088.89 1.831L40.049 45.712z"/></mask><g mask="url(#a)"><g filter="url(#b)"><ellipse cx="5.508" cy="14.704" fill="#eee6ff" rx="5.508" ry="14.704" transform="rotate(269.814 20.96 11.29)scale(-1 1)"/></g><g filter="url(#c)"><ellipse cx="10.399" cy="29.851" fill="#eee6ff" rx="10.399" ry="29.851" transform="rotate(89.814 -16.902 -8.275)scale(1 -1)"/></g><g filter="url(#d)"><ellipse cx="5.508" cy="30.487" fill="#8900ff" rx="5.508" ry="30.487" transform="rotate(89.814 -19.197 -7.127)scale(1 -1)"/></g><g filter="url(#e)"><ellipse cx="5.508" cy="30.599" fill="#8900ff" rx="5.508" ry="30.599" transform="rotate(89.814 -25.928 4.177)scale(1 -1)"/></g><g filter="url(#f)"><ellipse cx="5.508" cy="30.599" fill="#8900ff" rx="5.508" ry="30.599" transform="rotate(89.814 -25.738 5.52)scale(1 -1)"/></g><g filter="url(#g)"><ellipse cx="14.072" cy="22.078" fill="#eee6ff" rx="14.072" ry="22.078" transform="rotate(93.35 31.245 55.578)scale(-1 1)"/></g><g filter="url(#h)"><ellipse cx="3.47" cy="21.501" fill="#8900ff" rx="3.47" ry="21.501" transform="rotate(89.009 35.419 55.202)scale(-1 1)"/></g><g filter="url(#i)"><ellipse cx="3.47" cy="21.501" fill="#8900ff" rx="3.47" ry="21.501" transform="rotate(89.009 35.419 55.202)scale(-1 1)"/></g><g filter="url(#j)"><ellipse cx="14.592" cy="9.743" fill="#8900ff" rx="4.407" ry="29.108" transform="rotate(39.51 14.592 9.743)"/></g><g filter="url(#k)"><ellipse cx="61.728" cy="-5.321" fill="#8900ff" rx="4.407" ry="29.108" transform="rotate(37.892 61.728 -5.32)"/></g><g filter="url(#l)"><ellipse cx="55.618" cy="7.104" fill="#00c2ff" rx="5.971" ry="9.665" transform="rotate(37.892 55.618 7.104)"/></g><g filter="url(#m)"><ellipse cx="12.326" cy="39.103" fill="#8900ff" rx="4.407" ry="29.108" transform="rotate(37.892 12.326 39.103)"/></g><g filter="url(#n)"><ellipse cx="12.326" cy="39.103" fill="#8900ff" rx="4.407" ry="29.108" transform="rotate(37.892 12.326 39.103)"/></g><g filter="url(#o)"><ellipse cx="49.857" cy="30.678" fill="#8900ff" rx="4.407" ry="29.108" transform="rotate(37.892 49.857 30.678)"/></g><g filter="url(#p)"><ellipse cx="52.623" cy="33.171" fill="#00c2ff" rx="5.971" ry="15.297" transform="rotate(37.892 52.623 33.17)"/></g></g><path d="M6.919 0c-9.198 13.166-9.252 33.575 0 46.789h6.215c-9.25-13.214-9.196-33.623 0-46.789zm62.424 0h-6.215c9.198 13.166 9.252 33.575 0 46.789h6.215c9.25-13.214 9.196-33.623 0-46.789" class="parenthesis"/><defs><filter id="b" width="60.045" height="41.654" x="-5.564" y="16.92" color-interpolation-filters="sRGB" filterUnits="userSpaceOnUse"><feFlood flood-opacity="0" result="BackgroundImageFix"/><feBlend in="SourceGraphic" in2="BackgroundImageFix" result="shape"/><feGaussianBlur result="effect1_foregroundBlur_2002_17286" stdDeviation="7.659"/></filter><filter id="c" width="90.34" height="51.437" x="-40.407" y="-6.762" color-interpolation-filters="sRGB" filterUnits="userSpaceOnUse"><feFlood flood-opacity="0" result="BackgroundImageFix"/><feBlend in="SourceGraphic" in2="BackgroundImageFix" result="shape"/><feGaussianBlur result="effect1_foregroundBlur_2002_17286" stdDeviation="7.659"/></filter><filter id="d" width="79.355" height="29.4" x="-35.435" y="2.801" color-interpolation-filters="sRGB" filterUnits="userSpaceOnUse"><feFlood flood-opacity="0" result="BackgroundImageFix"/><feBlend in="SourceGraphic" in2="BackgroundImageFix" result="shape"/><feGaussianBlur result="effect1_foregroundBlur_2002_17286" stdDeviation="4.596"/></filter><filter id="e" width="79.579" height="29.4" x="-30.84" y="20.8" color-interpolation-filters="sRGB" filterUnits="userSpaceOnUse"><feFlood flood-opacity="0" result="BackgroundImageFix"/><feBlend in="SourceGraphic" in2="BackgroundImageFix" result="shape"/><feGaussianBlur result="effect1_foregroundBlur_2002_17286" stdDeviation="4.596"/></filter><filter id="f" width="79.579" height="29.4" x="-29.307" y="21.949" color-interpolation-filters="sRGB" filterUnits="usLine truncated
|
||||
|
Before Width: | Height: | Size: 8.5 KiB |
@@ -8,8 +8,6 @@ import { DATASETS, pathOf } from "../datasets";
|
||||
* No database is fetched on this route.
|
||||
*/
|
||||
export function Hub() {
|
||||
const current = DATASETS.filter((d) => !d.id.includes("old"));
|
||||
const archived = DATASETS.filter((d) => d.id.includes("old"));
|
||||
const base = import.meta.env.BASE_URL;
|
||||
|
||||
return (
|
||||
@@ -24,7 +22,7 @@ export function Hub() {
|
||||
|
||||
<main>
|
||||
<ul className="hub-list" role="list">
|
||||
{current.map((d) => (
|
||||
{DATASETS.map((d) => (
|
||||
<li key={d.id}>
|
||||
<a className="hub-link" href={pathOf(d, base)}>
|
||||
<span className="hub-label">{d.label}</span>
|
||||
@@ -33,20 +31,6 @@ export function Hub() {
|
||||
</li>
|
||||
))}
|
||||
</ul>
|
||||
|
||||
<section className="hub-archive" aria-labelledby="archive-heading">
|
||||
<h2 id="archive-heading">Phiên bản cũ của trang 2017</h2>
|
||||
<ul className="hub-list hub-list-compact" role="list">
|
||||
{archived.map((d) => (
|
||||
<li key={d.id}>
|
||||
<a className="hub-link" href={pathOf(d, base)}>
|
||||
<span className="hub-label">{d.label}</span>
|
||||
<span className="hub-blurb">{d.blurb}</span>
|
||||
</a>
|
||||
</li>
|
||||
))}
|
||||
</ul>
|
||||
</section>
|
||||
</main>
|
||||
|
||||
<footer>
|
||||
|
||||
+3
-5
@@ -24,7 +24,6 @@ const SUBTITLE = "Dữ liệu thí sinh toàn quốc · Hỗ trợ truy vấn SQ
|
||||
const CONTENT = {
|
||||
2016: {
|
||||
label: "Kỳ thi 2016",
|
||||
blurb: "877.461 thí sinh",
|
||||
title: "Tra cứu điểm thi THPT Quốc gia 2016",
|
||||
subtitle: SUBTITLE,
|
||||
source: "Bộ GD&ĐT",
|
||||
@@ -33,7 +32,6 @@ const CONTENT = {
|
||||
},
|
||||
2017: {
|
||||
label: "Kỳ thi 2017",
|
||||
blurb: "861.068 thí sinh",
|
||||
title: "Tra cứu điểm thi THPT Quốc gia 2017",
|
||||
subtitle: SUBTITLE,
|
||||
source: "baotintuc.vn",
|
||||
@@ -60,12 +58,12 @@ for (const id of Object.keys(CONTENT)) {
|
||||
export const DATASETS = registry.datasets.map((d) => ({
|
||||
id: d.id,
|
||||
dbSizeMb: d.dbSizeMb,
|
||||
// Derived, not written twice: the candidate count the hub shows is the same
|
||||
// number the assembler enforces, so the two cannot drift apart.
|
||||
blurb: `${d.expectedRows.toLocaleString("vi-VN")} thí sinh`,
|
||||
...CONTENT[d.id],
|
||||
}));
|
||||
|
||||
/** Dataset IDs in build order. */
|
||||
export const DATASET_IDS = DATASETS.map((d) => d.id);
|
||||
|
||||
/** Site path for a dataset, e.g. pathOf(d, "/thptqg/") → "/thptqg/2017/". */
|
||||
export function pathOf(dataset, base) {
|
||||
return `${base}${dataset.id}/`;
|
||||
|
||||
+19
-23
@@ -1,7 +1,7 @@
|
||||
/**
|
||||
* The 16 subject columns of the canonical schema, in display order.
|
||||
*
|
||||
* Mirrors `SCORE_FIELDS` in parser/internal/schema/schema.go. Previously this list was
|
||||
* Mirrors `ScoreFields` in parser/internal/schema/schema.go. Previously this list was
|
||||
* maintained separately in score-table.jsx and student-detail.jsx, which is how
|
||||
* they drifted out of sync with each other and with the database.
|
||||
*
|
||||
@@ -11,29 +11,29 @@
|
||||
*/
|
||||
|
||||
export const SUBJECTS = [
|
||||
{ key: "toan", label: "Toán", short: "Toán" },
|
||||
{ key: "ngu_van", label: "Ngữ văn", short: "Văn" },
|
||||
{ key: "vat_ly", label: "Vật lí", short: "Lý" },
|
||||
{ key: "hoa_hoc", label: "Hóa học", short: "Hóa" },
|
||||
{ key: "sinh_hoc", label: "Sinh học", short: "Sinh" },
|
||||
{ key: "khtn", label: "KHTN", short: "KHTN" },
|
||||
{ key: "lich_su", label: "Lịch sử", short: "Sử" },
|
||||
{ key: "dia_ly", label: "Địa lí", short: "Địa" },
|
||||
{ key: "gdcd", label: "GDCD", short: "GDCD" },
|
||||
{ key: "khxh", label: "KHXH", short: "KHXH" },
|
||||
{ key: "tieng_anh", label: "Tiếng Anh", short: "T.Anh" },
|
||||
{ key: "tieng_phap", label: "Tiếng Pháp", short: "T.Pháp" },
|
||||
{ key: "tieng_nga", label: "Tiếng Nga", short: "T.Nga" },
|
||||
{ key: "tieng_duc", label: "Tiếng Đức", short: "T.Đức" },
|
||||
{ key: "tieng_nhat", label: "Tiếng Nhật", short: "T.Nhật" },
|
||||
{ key: "tieng_trung", label: "Tiếng Trung", short: "T.Trung" },
|
||||
{ key: "toan", label: "Toán" },
|
||||
{ key: "ngu_van", label: "Ngữ văn" },
|
||||
{ key: "vat_ly", label: "Vật lí" },
|
||||
{ key: "hoa_hoc", label: "Hóa học" },
|
||||
{ key: "sinh_hoc", label: "Sinh học" },
|
||||
{ key: "khtn", label: "KHTN" },
|
||||
{ key: "lich_su", label: "Lịch sử" },
|
||||
{ key: "dia_ly", label: "Địa lí" },
|
||||
{ key: "gdcd", label: "GDCD" },
|
||||
{ key: "khxh", label: "KHXH" },
|
||||
{ key: "tieng_anh", label: "Tiếng Anh" },
|
||||
{ key: "tieng_phap", label: "Tiếng Pháp" },
|
||||
{ key: "tieng_nga", label: "Tiếng Nga" },
|
||||
{ key: "tieng_duc", label: "Tiếng Đức" },
|
||||
{ key: "tieng_nhat", label: "Tiếng Nhật" },
|
||||
{ key: "tieng_trung", label: "Tiếng Trung" },
|
||||
];
|
||||
|
||||
/**
|
||||
* Non-score columns worth showing in the results table.
|
||||
*
|
||||
* Only the 2016 dataset populates these; they are NULL throughout the 2017
|
||||
* datasets and get hidden by the same all-NULL filter that hides unused
|
||||
* Only the 2016 dataset populates these; they are NULL throughout 2017
|
||||
* and get hidden by the same all-NULL filter that hides unused
|
||||
* subjects, so no per-dataset conditional is needed.
|
||||
*/
|
||||
export const IDENTITY_COLUMNS = [
|
||||
@@ -41,10 +41,6 @@ export const IDENTITY_COLUMNS = [
|
||||
{ key: "gioi_tinh", label: "GT" },
|
||||
];
|
||||
|
||||
export const SUBJECT_LABELS = Object.fromEntries(
|
||||
SUBJECTS.map((s) => [s.key, s.label]),
|
||||
);
|
||||
|
||||
/** True when at least one row carries a value for `key`. */
|
||||
export function hasAnyValue(rows, key) {
|
||||
return rows.some((row) => row[key] !== null && row[key] !== undefined);
|
||||
|
||||
+4
-4
@@ -10,8 +10,8 @@ import react from "@vitejs/plugin-react";
|
||||
// path, so every URL is a real static file and no SPA 404-fallback is needed —
|
||||
// and the existing ?q= deep links keep working, which that fallback would break.
|
||||
//
|
||||
// publicDir holds only the gzipped databases, staged there by
|
||||
// parser/scripts/build-db.js. Nothing uncompressed is ever placed in it.
|
||||
// publicDir holds only the gzipped databases, staged there by the assembler
|
||||
// (assembler/internal/databases). Nothing uncompressed is ever placed in it.
|
||||
//
|
||||
// It sits at the repository root rather than inside this workspace because
|
||||
// parser writes it, so the path has to climb out of web/. Vite resolves
|
||||
@@ -19,8 +19,8 @@ import react from "@vitejs/plugin-react";
|
||||
//
|
||||
// outDir is left at its default, so the build lands in web/dist and this
|
||||
// workspace stays self-contained. The Pages artifact (_site) is assembled at
|
||||
// the repository root by web/scripts/assemble-site.js, since that is where the
|
||||
// deploy action uploads from.
|
||||
// the repository root by the assembler (assembler/internal/site), since that is
|
||||
// where the deploy action uploads from.
|
||||
export default defineConfig({
|
||||
plugins: [react()],
|
||||
base: "/thptqg/",
|
||||
|
||||
Reference in new issue
Block a user