Files
thptqg/parser/scripts/differential-parity.mjs
T
tiennm99 c359a0b444 refactor: one directory per pipeline stage, and an assembler to drive them
The repository now reads as the pipeline it is: crawler fetches, parser
converts, assembler verifies and publishes, with data/ and web/ as the stores
they hand work through. go-parser is renamed parser now that there is no other.

The assembler replaces build-db.js and assemble-site.js. It compiles the
parser, builds and verifies each database, compresses it, runs the Vite build
and assembles _site — one command, and the only place that knows the order.

It also closes a real hole: nothing previously asserted that a database reached
the site. An empty staging directory assembled happily, so every page rendered,
every query 404d and CI stayed green. The row-count and size guards could not
catch that, since they only run when a database was built at all.

Removing Node from the root forced the dataset list out of web/src/datasets.js,
which the assembler cannot import. datasets.json is now the registry both sides
read — JSON because Go and the browser both parse it without a dependency —
while presentation stays in the web app, keyed by id and cross-checked against
the registry so a half-added dataset fails instead of half-working.

Guards verified by making each one fail: a missing database, and an expected
row count one higher than the truth.
2026-08-13 22:50:05 +07:00

208 lines
7.4 KiB
JavaScript

#!/usr/bin/env node
// Differential parity gate: proves the Go parser's databases are logically
// equivalent to the Rust parser's, across all four datasets.
//
// This is the entire safety argument for the migration. Unit and golden tests
// use synthetic fixtures; this is the only check against all 418 MB of real
// input.
//
// Deliberately NOT built on parser/scripts/verify-parity.js. That script
// compares a schema-shape change against a frozen baseline: its core check is
// "new columns must be all-NULL except an approved allowlist", which is vacuous
// here because both databases have the identical 22 columns, and its
// APPROVED_RECOVERY guard would emit seven spurious failures on a perfectly
// correct port. It also passes silently when a dataset is missing from both
// inputs. Leave it alone; it remains valid for the historical check it was
// written for.
//
// Uses node:sqlite, already the repo's only SQLite client — no new dependency,
// and no sqlite3 CLI (there isn't one on this box).
//
// Usage:
// node go-parser/scripts/differential-parity.mjs \
// --rust /tmp/rust-{id}.db --go /tmp/go-{id}.db \
// [--rust-stdout /tmp/rust-{id}.stdout --go-stdout /tmp/go-{id}.stdout]
//
// {id} is substituted per dataset. Exits non-zero on any mismatch.
import { createHash } from "node:crypto";
import { existsSync, readFileSync } from "node:fs";
import { DatabaseSync } from "node:sqlite";
const DATASETS = ["2016", "2017", "2017-old", "2017-old2"];
function arg(name, fallback) {
const i = process.argv.indexOf(name);
return i >= 0 && process.argv[i + 1] ? process.argv[i + 1] : fallback;
}
const rustPattern = arg("--rust", "/tmp/rust-{id}.db");
const goPattern = arg("--go", "/tmp/go-{id}.db");
const rustOutPattern = arg("--rust-stdout", "/tmp/rust-{id}.stdout");
const goOutPattern = arg("--go-stdout", "/tmp/go-{id}.stdout");
const expand = (p, id) => p.replace("{id}", id);
const failures = [];
const fail = (ds, msg) => failures.push(`${ds}: ${msg}`);
/** Column names in declaration order, from the table itself. */
function columns(db) {
return db.prepare("PRAGMA table_info(student)").all().map((r) => r.name);
}
/** PRAGMA metadata, normalised to a comparable string. */
function schemaSignature(db) {
const cols = db
.prepare("PRAGMA table_info(student)")
.all()
.map((r) => `${r.cid}:${r.name}:${r.type}:${r.notnull}:${r.pk}`)
.join("|");
const idx = db
.prepare("PRAGMA index_list(student)")
.all()
.map((r) => `${r.name}:${r.unique}:${r.partial}`)
.sort()
.join("|");
return `${cols}\n${idx}`;
}
/**
* Serialise one value so the hash is stable across drivers.
*
* Rust and Go embed different SQLite versions, so byte-identical files are
* impossible by construction; what must match is the logical content. NULL gets
* an explicit sentinel that no real value can collide with, and REALs are
* rendered with a fixed rule rather than whatever each driver's default
* formatting happens to be.
*/
function ser(v) {
if (v === null || v === undefined) return "NULL";
if (typeof v === "number") return Number.isInteger(v) ? v.toFixed(1) : String(v);
return String(v);
}
/**
* Rolling SHA-256 over every row, ordered by primary key.
*
* Streamed rather than materialised: 877k rows x 22 columns would otherwise be
* a large amount of memory for no benefit.
*/
function tableHash(db, cols) {
const h = createHash("sha256");
const stmt = db.prepare(`SELECT * FROM student ORDER BY so_bao_danh`);
let n = 0;
for (const row of stmt.iterate()) {
h.update(cols.map((c) => ser(row[c])).join(""));
h.update("");
n++;
}
return { hash: h.digest("hex"), rows: n };
}
/** First N rows whose serialisation differs, for diagnosis. */
function firstDifferences(rdb, gdb, cols, limit = 20) {
const out = [];
const ri = rdb.prepare("SELECT * FROM student ORDER BY so_bao_danh").iterate();
const gi = gdb.prepare("SELECT * FROM student ORDER BY so_bao_danh").iterate();
for (;;) {
const r = ri.next();
const g = gi.next();
if (r.done || g.done) break;
for (const c of cols) {
if (ser(r.value[c]) !== ser(g.value[c])) {
out.push(
` so_bao_danh=${r.value.so_bao_danh} ${c}: rust=${JSON.stringify(r.value[c])} go=${JSON.stringify(g.value[c])}`
);
break;
}
}
if (out.length >= limit) break;
}
return out;
}
/** stdout comparison, ignoring only the output path in the header line. */
function compareStdout(ds) {
const rp = expand(rustOutPattern, ds);
const gp = expand(goOutPattern, ds);
if (!existsSync(rp) || !existsSync(gp)) {
console.log(` stdout : skipped (capture not found)`);
return;
}
const norm = (p) =>
readFileSync(p, "utf8").replace(/→ \S+/, "→ DB");
if (norm(rp) === norm(gp)) {
console.log(` stdout : identical`);
} else {
fail(ds, "stdout differs");
console.log(` stdout : *** DIFFERS ***`);
}
}
console.log("=== differential parity: Rust vs Go ===\n");
for (const ds of DATASETS) {
const rp = expand(rustPattern, ds);
const gp = expand(goPattern, ds);
console.log(`--- ${ds} ---`);
// Fail loudly on a missing dataset rather than skipping it: silently comparing
// three of four datasets is how a gate passes without proving anything.
if (!existsSync(rp)) { fail(ds, `rust db missing: ${rp}`); console.log(` MISSING ${rp}`); continue; }
if (!existsSync(gp)) { fail(ds, `go db missing: ${gp}`); console.log(` MISSING ${gp}`); continue; }
const rdb = new DatabaseSync(rp, { readOnly: true });
const gdb = new DatabaseSync(gp, { readOnly: true });
const rSig = schemaSignature(rdb);
const gSig = schemaSignature(gdb);
if (rSig !== gSig) {
fail(ds, "schema/index metadata differs");
console.log(` schema : *** DIFFERS ***`);
} else {
console.log(` schema : identical (table_info + index_list)`);
}
const cols = columns(rdb);
const rCount = rdb.prepare("SELECT COUNT(*) c FROM student").get().c;
const gCount = gdb.prepare("SELECT COUNT(*) c FROM student").get().c;
if (rCount !== gCount) fail(ds, `row count ${rCount} vs ${gCount}`);
console.log(` rows : ${rCount} vs ${gCount} ${rCount === gCount ? "OK" : "*** MISMATCH ***"}`);
let colMismatch = 0;
for (const c of cols) {
const r = rdb.prepare(`SELECT COUNT(${c}) n FROM student`).get().n;
const g = gdb.prepare(`SELECT COUNT(${c}) n FROM student`).get().n;
if (r !== g) {
colMismatch++;
fail(ds, `non-NULL count for ${c}: ${r} vs ${g}`);
console.log(` ${c}: ${r} vs ${g} *** MISMATCH ***`);
}
}
console.log(` non-NULL : all ${cols.length} columns ${colMismatch === 0 ? "OK" : `${colMismatch} MISMATCHED`}`);
const rh = tableHash(rdb, cols);
const gh = tableHash(gdb, cols);
if (rh.hash !== gh.hash) {
fail(ds, `full-table hash differs`);
console.log(` full-table : *** DIFFERS ***\n rust=${rh.hash.slice(0, 32)}\n go =${gh.hash.slice(0, 32)}`);
for (const line of firstDifferences(rdb, gdb, cols)) console.log(line);
} else {
console.log(` full-table : identical sha256 ${rh.hash.slice(0, 16)} over ${rh.rows} rows`);
}
compareStdout(ds);
rdb.close();
gdb.close();
console.log("");
}
if (failures.length) {
console.log(`PARITY FAILED — ${failures.length} problem(s):`);
for (const f of failures) console.log(` - ${f}`);
process.exit(1);
}
console.log("PARITY OK — all 4 datasets logically equivalent (rows, per-column non-NULL, full-table hash, schema, stdout).");