Files
thptqg/parser/scripts/build-db.js
T
tiennm99 0eb174721d refactor(parser): reimplement the parser in Go alongside the Rust crate
Adds go-parser/, a Go reimplementation of the xlsxread parser, verified
byte-for-byte against the Rust original before any cutover.

Reader fidelity is exact across all 299 input files: the canonical cell dump
of every sheet matches calamine's, locked in as a test against a committed
hash oracle. Reaching that required replacing extrame/xls, which corrupted
69% of cells and dropped a further 28% on the BIFF corpus, with pbnjay/grate;
correcting excelize's number-format application and trailing-cell trimming;
restoring carriage returns that XML line-ending normalisation strips from
2,233 ten_cum_thi values; and gating numeric re-rendering on cell type so
shared strings that merely look numeric keep their leading zeros.

The differential gate compares both parsers over all four datasets:
3,265,641 rows with identical full-table SHA-256, identical per-column
non-NULL counts, identical schema metadata and identical stdout.

Config moves from TOML to YAML for both parsers, so they keep reading the
same files and the gate stays meaningful. Verified by rebuilding 2016 and
2017-old2 with Rust under the new configs and matching the recorded counts.

build-db.js now refuses to publish a database whose row count does not match
the known figure, closing a path where an under-producing parser could ship a
truncated public dataset with green CI. The deploy workflow gains a
pull_request trigger and guards deploy to main, so branch verification can no
longer publish to production.
2026-08-13 20:27:44 +07:00

67 lines
2.0 KiB
JavaScript

#!/usr/bin/env node
/**
* Build the SQLite database for one or all datasets, then gzip it.
*
* Replaces the six per-dataset npm scripts the two old projects carried. The
* dataset list comes from src/datasets.js so it is written in exactly one place.
*
* Output goes to .build/public/db/ — the directory Vite copies as its publicDir.
* Only the .gz survives: shipping a 100+ MB uncompressed database is made
* structurally impossible rather than left to a cleanup step.
*
* Usage:
* node parser/scripts/build-db.js # all four datasets
* node parser/scripts/build-db.js 2017-old # just one
*/
import { execFileSync } from "node:child_process";
import { mkdirSync, rmSync, existsSync } from "node:fs";
import { dirname, resolve } from "node:path";
import { fileURLToPath } from "node:url";
import { DATASET_IDS } from "../../src/datasets.js";
const ROOT = resolve(dirname(fileURLToPath(import.meta.url)), "../..");
const BIN = resolve(ROOT, "parser/target/release/xlsxread");
const OUT_DIR = resolve(ROOT, ".build/public/db");
const requested = process.argv.slice(2);
const unknown = requested.filter((id) => !DATASET_IDS.includes(id));
if (unknown.length) {
console.error(`unknown dataset(s): ${unknown.join(", ")}`);
console.error(`known: ${DATASET_IDS.join(", ")}`);
process.exit(2);
}
const targets = requested.length ? requested : DATASET_IDS;
if (!existsSync(BIN)) {
console.error(`parser binary not found at ${BIN}`);
console.error("run: npm run build:rust");
process.exit(1);
}
mkdirSync(OUT_DIR, { recursive: true });
for (const id of targets) {
const db = resolve(OUT_DIR, `${id}.db`);
execFileSync(
BIN,
[
"build",
"--schema",
resolve(ROOT, `parser/configs/${id}.yml`),
"--input",
resolve(ROOT, `data/${id}`),
"--output",
db,
],
{ stdio: "inherit" },
);
// -9 without -k: the raw .db must not reach the published artifact.
rmSync(`${db}.gz`, { force: true });
execFileSync("gzip", ["-9", db], { stdio: "inherit" });
console.log(` → db/${id}.db.gz\n`);
}