feat: replace xlsx (SheetJS) build pipeline with Rust xlsxread CLI (#1)

* feat(xlsxread): vendor Rust binary cloned from thptqg2017

Copies the xlsxread Rust crate from thptqg2017@8b4a755 (chore/xlsxread-rust).
Adds format_detect_2016 module with per-file column-layout auto-detection
mirroring detectFormat() in scripts/build-database.js (lines 63-87):
  - separate-scores: SBD/HOTEN/TOAN... fixed columns (dhhanghai files)
  - mapped: header-derived SOBAODANH|SBD + DIEM_THI dynamic indices
  - default: positional 6-col layout (no header)

Extends ParsedRow with ten_cum_thi and gioi_tinh fields.
Adds SCORE_FIELDS_2016 (12 cols: tieng_duc/tieng_nhat; no khtn/khxh/tieng_nga).
Adds thptqg2016-data.toml config with 18-column schema and format_detection flag.
58 tests pass (50 unit + 8 integration), 0 failures.

* feat(build): wire build:db to xlsxread CLI

Replaces the Node.js build:db script with the xlsxread Rust binary.
Adds build:rust script for the cargo compile step in isolation.

* chore: remove deprecated build-database.js

Superseded by the xlsxread Rust binary. All 119 source files (4 .xls +
115 .xlsx) are now processed by xlsxread with per-file format detection.

* ci: build xlsxread before running database build job

Adds dtolnay/rust-toolchain@stable and Swatinem/rust-cache@v2 steps
before the xlsxread build and database generation steps. Node/pnpm
steps now follow the Rust build rather than preceding it.

* chore: add Rust build artifacts to .gitignore

* docs: update README build instructions for xlsxread pipeline

* chore(deps): drop xlsx and better-sqlite3 from package.json and lockfile
This commit is contained in:
tiennm99 authored and GitHub committed 2026-05-19 16:33:20 +07:00
1 parent 99465fda59
commit 6b21c625ac
27 files changed
+4422 -315

No files matched your search

+16 -3
View File
@@ -20,6 +20,22 @@ jobs:
steps:
- uses: actions/checkout@v4
- uses: dtolnay/rust-toolchain@stable
- uses: Swatinem/rust-cache@v2
with:
workspaces: tools/xlsxread
- name: Build xlsxread binary
run: cargo build --release --manifest-path tools/xlsxread/Cargo.toml
- name: Build database
run: >
./tools/xlsxread/target/release/xlsxread build
--schema tools/xlsxread/configs/thptqg2016-data.toml
--input data
--output public/thptqg2016.db
- uses: pnpm/action-setup@v4
- uses: actions/setup-node@v4
@@ -30,9 +46,6 @@ jobs:
- name: Install dependencies
run: pnpm install --frozen-lockfile
- name: Build database
run: pnpm build:db
- name: Compress database
run: gzip -k -9 public/thptqg2016.db
+3
View File
@@ -141,3 +141,6 @@ vite.config.ts.timestamp-*
# Generated database files
public/*.db
public/*.db.gz
# Rust build artifacts
tools/xlsxread/target/
+22 -7
View File
@@ -19,18 +19,33 @@ Fully static app running entirely in the browser (SQLite via `sql.js`). No backe
## Development
```bash
npm install
npm run build:db # Parse data/*.xlsx → public/thptqg2016.db
npm run dev # Vite dev server
npm run build # Production bundle → dist/
npm run lint # ESLint
# Build the SQLite database from source Excel files (requires Rust stable)
pnpm run build:db
# Or build in two steps:
pnpm run build:rust # compile the xlsxread binary
./tools/xlsxread/target/release/xlsxread build \
--schema tools/xlsxread/configs/thptqg2016-data.toml \
--input data \
--output public/thptqg2016.db
pnpm run dev # Vite dev server
pnpm run build # Production bundle → dist/
pnpm run lint # ESLint
```
The GitHub Actions workflow (`.github/workflows/deploy.yml`) builds the DB, gzips it, and deploys to GitHub Pages on every push to `main`.
The database is built by the `xlsxread` Rust binary (`tools/xlsxread/`), which
reads the 119 mixed `.xls`/`.xlsx` files and auto-detects the column layout per
file (`separate-scores`, `mapped`, or positional default). No Node.js Excel
library is required at build time.
The GitHub Actions workflow (`.github/workflows/deploy.yml`) compiles
`xlsxread`, builds the DB, gzips it, and deploys to GitHub Pages on every push
to `main`.
## Tech stack
React 19 · Vite · sql.js (WASM) · better-sqlite3 (build-time only) · GitHub Pages
React 19 · Vite · sql.js (WASM) · xlsxread (Rust, build-time) · GitHub Pages
## Documentation
+3 -4
View File
@@ -6,7 +6,8 @@
"type": "module",
"description": "Tra cứu điểm thi THPT QG 2016 - Hỗ trợ truy vấn SQL tùy chỉnh",
"scripts": {
"build:db": "node scripts/build-database.js",
"build:rust": "cargo build --release --manifest-path tools/xlsxread/Cargo.toml",
"build:db": "cargo build --release --manifest-path tools/xlsxread/Cargo.toml && ./tools/xlsxread/target/release/xlsxread build --schema tools/xlsxread/configs/thptqg2016-data.toml --input data --output public/thptqg2016.db",
"dev": "vite",
"build": "vite build",
"preview": "vite preview",
@@ -27,12 +28,10 @@
"@types/react": "^19.2.14",
"@types/react-dom": "^19.2.3",
"@vitejs/plugin-react": "^6.0.1",
"better-sqlite3": "^12.8.0",
"eslint": "^9.39.4",
"eslint-plugin-react-hooks": "^7.0.1",
"eslint-plugin-react-refresh": "^0.5.2",
"globals": "^17.4.0",
"vite": "^8.0.4",
"xlsx": "^0.18.5"
"vite": "^8.0.4"
}
}
-30
View File
@@ -30,9 +30,6 @@ importers:
'@vitejs/plugin-react':
specifier: ^6.0.1
version: 6.0.1(vite@8.0.12)
better-sqlite3:
specifier: ^12.8.0
version: 12.9.0
eslint:
specifier: ^9.39.4
version: 9.39.4
@@ -48,9 +45,6 @@ importers:
vite:
specifier: ^8.0.4
version: 8.0.12
xlsx:
specifier: ^0.18.5
version: 0.18.5
packages:
@@ -379,10 +373,6 @@ packages:
engines: {node: '>=6.0.0'}
hasBin: true
better-sqlite3@12.9.0:
resolution: {integrity: sha512-wqUv4Gm3toFpHDQmaKD4QhZm3g1DjUBI0yzS4UBl6lElUmXFYdTQmmEDpAFa5o8FiFiymURypEnfVHzILKaxqQ==}
engines: {node: 20.x || 22.x || 23.x || 24.x || 25.x}
bindings@1.5.0:
resolution: {integrity: sha512-p2q/t/mhvuOj/UeLlV6566GD/guowlr0hHxClI0W9m7MWYkL1F0hLo+0Aexs9HSPCtR1SXQ0TD3MMKrXZajbiQ==}
@@ -1034,11 +1024,6 @@ packages:
wrappy@1.0.2:
resolution: {integrity: sha512-l4Sp/DRseor9wL6EvV2+TuQn63dMkPjZ/sp9XkghTEbV9KlPS1xUsZ3u7/IQO4wxtcFB4bgpQPRcR3QCvezPcQ==}
xlsx@0.18.5:
resolution: {integrity: sha512-dmg3LCjBPHZnQp5/F/+nnTa+miPJxUXB6vtk42YjBBKayDNagxGEeIdWApkYPOf3Z3pm3k62Knjzp7lMeTEtFQ==}
engines: {node: '>=0.8'}
hasBin: true
yallist@3.1.1:
resolution: {integrity: sha512-a4UGQaWPH59mOXUYnAG2ewncQS4i4F43Tv3JoAM+s2VDAmS9NsK8GpDMLrCHPksFT7h3K6TOoUNn2pb7RoXx4g==}
@@ -1365,11 +1350,6 @@ snapshots:
baseline-browser-mapping@2.10.29: {}
better-sqlite3@12.9.0:
dependencies:
bindings: 1.5.0
prebuild-install: 7.1.3
bindings@1.5.0:
dependencies:
file-uri-to-path: 1.0.0
@@ -1942,16 +1922,6 @@ snapshots:
wrappy@1.0.2: {}
xlsx@0.18.5:
dependencies:
adler-32: 1.3.1
cfb: 1.2.2
codepage: 1.15.0
crc-32: 1.2.2
ssf: 0.11.2
wmf: 1.0.2
word: 0.3.0
yallist@3.1.1: {}
yocto-queue@0.1.0: {}
-271
View File
@@ -1,271 +0,0 @@
import XLSX from "xlsx";
import Database from "better-sqlite3";
import fs from "fs";
import path from "path";
import { fileURLToPath } from "url";
const __dirname = path.dirname(fileURLToPath(import.meta.url));
const DATA_DIR = path.join(__dirname, "..", "data");
const DB_PATH = path.join(__dirname, "..", "public", "thptqg2016.db");
// Score patterns for the DIEM_THI string format
const SCORE_PATTERNS = {
toan: /Toán:\s*([\d.]+)/,
ngu_van: /Ngữ văn:\s*([\d.]+)/,
vat_ly: /Vật lí:\s*([\d.]+)/,
hoa_hoc: /Hóa học:\s*([\d.]+)/,
sinh_hoc: /Sinh học:\s*([\d.]+)/,
lich_su: /Lịch sử:\s*([\d.]+)/,
dia_ly: /Địa lí:\s*([\d.]+)/,
tieng_anh: /Tiếng Anh:\s*([\d.]+)/,
tieng_phap: /Tiếng Pháp:\s*([\d.]+)/,
tieng_duc: /Tiếng Đức:\s*([\d.]+)/,
tieng_nhat: /Tiếng Nhật:\s*([\d.]+)/,
tieng_trung: /Tiếng Trung:\s*([\d.]+)/,
};
const ALL_SCORE_FIELDS = Object.keys(SCORE_PATTERNS);
// Strip Vietnamese diacritics: "NGUYỄN BŨU LỘC" → "nguyen buu loc"
function toAscii(str) {
return str
.normalize("NFD")
.replace(/[\u0300-\u036f]/g, "")
.replace(/đ/g, "d")
.replace(/Đ/g, "D")
.toLowerCase();
}
// Parse score text "Toán: 3.75 Ngữ văn: 5.00 ..." into { toan: 3.75, ... }
function parseScoreString(diemThi) {
const scores = {};
for (const [field, pattern] of Object.entries(SCORE_PATTERNS)) {
const match = diemThi.match(pattern);
if (match) scores[field] = parseFloat(match[1]);
}
return scores;
}
// Detect header row by checking for known column names
const KNOWN_HEADERS = new Set([
"SOBAODANH", "SBD", "HO_TEN", "HOTEN", "HỌ TÊN",
"NGAY_SINH", "TEN_CUMTHI", "GIOI_TINH", "DIEM_THI", "STT",
"TOAN", "VAN", "LY", "HOA", "SINH ", "SU", "DIA",
]);
function isHeaderRow(row) {
if (!row || row.length < 2) return false;
const first = String(row[0] || "").trim().toUpperCase();
return KNOWN_HEADERS.has(first);
}
// Detect which format a file uses based on its header row
function detectFormat(headerRow) {
if (!headerRow) return null;
const cols = headerRow.map((c) => String(c || "").trim().toUpperCase());
// Format: SBD, HOTEN, TOAN, VAN, LY, HOA, SINH, SU, DIA, ...
if (cols[0] === "SBD" && cols[2] === "TOAN") return "separate-scores";
// Build a column index map for flexible column ordering
const map = {};
for (let i = 0; i < cols.length; i++) {
const c = cols[i];
if (c === "SOBAODANH" || c === "SBD") map.sbd = i;
else if (c === "HO_TEN" || c === "HOTEN" || c === "HỌ TÊN") map.ho_ten = i;
else if (c === "NGAY_SINH") map.ngay_sinh = i;
else if (c === "TEN_CUMTHI") map.ten_cum_thi = i;
else if (c === "GIOI_TINH") map.gioi_tinh = i;
else if (c === "DIEM_THI") map.diem_thi = i;
}
if (map.sbd !== undefined && map.diem_thi !== undefined) {
return { type: "mapped", map };
}
return null;
}
// Process a file with separate score columns (dhhanghai format)
function processSeparateScoresRow(row) {
const sbd = String(row[0] || "").trim();
const hoTen = String(row[1] || "").trim();
if (!sbd || !hoTen) return null;
return {
so_bao_danh: sbd,
ho_ten: hoTen,
ho_ten_ascii: toAscii(hoTen),
ngay_sinh: null,
ten_cum_thi: null,
gioi_tinh: null,
toan: parseFloat(row[2]) || null,
ngu_van: parseFloat(row[3]) || null,
vat_ly: parseFloat(row[4]) || null,
hoa_hoc: parseFloat(row[5]) || null,
sinh_hoc: parseFloat(row[6]) || null,
lich_su: parseFloat(row[7]) || null,
dia_ly: parseFloat(row[8]) || null,
// row[9]=NGOAINGUTN, row[10]=NGOAINGUTL, row[11]=NGOAINGU (total)
tieng_anh: parseFloat(row[11]) || null,
tieng_phap: null,
tieng_duc: null,
tieng_nhat: null,
tieng_trung: null,
};
}
// Process a row using the column map
function processMappedRow(row, map) {
const sbd = String(row[map.sbd] || "").trim();
const hoTen = String(row[map.ho_ten] || "").trim();
if (!sbd || !hoTen) return null;
// Skip leaked header rows
const sbdUpper = sbd.toUpperCase();
if (KNOWN_HEADERS.has(sbdUpper) || KNOWN_HEADERS.has(hoTen.toUpperCase())) return null;
const ngaySinh = map.ngay_sinh !== undefined ? String(row[map.ngay_sinh] || "").trim() : null;
const tenCumThi = map.ten_cum_thi !== undefined ? String(row[map.ten_cum_thi] || "").trim() : null;
const rawGioiTinh = map.gioi_tinh !== undefined ? String(row[map.gioi_tinh] || "").trim() : null;
// Normalize gender: only accept "Nam" or "Nữ"
const gioiTinh = (rawGioiTinh === "Nam" || rawGioiTinh === "Nữ") ? rawGioiTinh : null;
const diemThi = map.diem_thi !== undefined ? String(row[map.diem_thi] || "") : "";
const scores = parseScoreString(diemThi);
return {
so_bao_danh: sbd,
ho_ten: hoTen,
ho_ten_ascii: toAscii(hoTen),
ngay_sinh: ngaySinh || null,
ten_cum_thi: tenCumThi || null,
gioi_tinh: gioiTinh || null,
...Object.fromEntries(ALL_SCORE_FIELDS.map((f) => [f, scores[f] ?? null])),
};
}
// Standard 6-column format without header: SBD, HO_TEN, NGAY_SINH, TEN_CUMTHI, GIOI_TINH, DIEM_THI
const DEFAULT_MAP = {
sbd: 0, ho_ten: 1, ngay_sinh: 2, ten_cum_thi: 3, gioi_tinh: 4, diem_thi: 5,
};
function main() {
fs.mkdirSync(path.dirname(DB_PATH), { recursive: true });
if (fs.existsSync(DB_PATH)) fs.unlinkSync(DB_PATH);
const db = new Database(DB_PATH);
db.exec(`
CREATE TABLE student (
so_bao_danh TEXT PRIMARY KEY,
ho_ten TEXT NOT NULL,
ho_ten_ascii TEXT NOT NULL,
ngay_sinh TEXT,
ten_cum_thi TEXT,
gioi_tinh TEXT,
toan REAL,
ngu_van REAL,
vat_ly REAL,
hoa_hoc REAL,
sinh_hoc REAL,
lich_su REAL,
dia_ly REAL,
tieng_anh REAL,
tieng_phap REAL,
tieng_duc REAL,
tieng_nhat REAL,
tieng_trung REAL
);
CREATE INDEX idx_ho_ten ON student(ho_ten);
CREATE INDEX idx_ho_ten_ascii ON student(ho_ten_ascii);
CREATE INDEX idx_ten_cum_thi ON student(ten_cum_thi);
`);
const insert = db.prepare(`
INSERT OR REPLACE INTO student
(so_bao_danh, ho_ten, ho_ten_ascii, ngay_sinh, ten_cum_thi, gioi_tinh,
toan, ngu_van, vat_ly, hoa_hoc, sinh_hoc, lich_su, dia_ly,
tieng_anh, tieng_phap, tieng_duc, tieng_nhat, tieng_trung)
VALUES
(@so_bao_danh, @ho_ten, @ho_ten_ascii, @ngay_sinh, @ten_cum_thi, @gioi_tinh,
@toan, @ngu_van, @vat_ly, @hoa_hoc, @sinh_hoc, @lich_su, @dia_ly,
@tieng_anh, @tieng_phap, @tieng_duc, @tieng_nhat, @tieng_trung)
`);
// Collect all Excel files (.xlsx and .xls)
const files = fs.readdirSync(DATA_DIR)
.filter((f) => f.endsWith(".xlsx") || f.endsWith(".xls"))
.map((f) => path.join(DATA_DIR, f));
let totalRows = 0;
let errorCount = 0;
const insertAll = db.transaction((files) => {
for (const file of files) {
const basename = path.basename(file);
let fileRows = 0;
try {
const wb = XLSX.readFile(file);
const ws = wb.Sheets[wb.SheetNames[0]];
const rows = XLSX.utils.sheet_to_json(ws, { header: 1 });
if (rows.length === 0) continue;
let startRow = 0;
let format = null;
if (isHeaderRow(rows[0])) {
format = detectFormat(rows[0]);
startRow = 1;
}
for (let i = startRow; i < rows.length; i++) {
const row = rows[i];
if (!row || row.length < 2) continue;
try {
let record;
if (format === "separate-scores") {
record = processSeparateScoresRow(row);
} else if (format && format.type === "mapped") {
record = processMappedRow(row, format.map);
} else {
// No header or unrecognized: assume standard 6-column order
record = processMappedRow(row, DEFAULT_MAP);
}
if (!record) continue;
insert.run(record);
fileRows++;
} catch {
errorCount++;
}
}
} catch (err) {
console.error(`Failed to read ${basename}: ${err.message}`);
}
totalRows += fileRows;
console.log(` ${basename}: ${fileRows} rows`);
}
});
console.log(`Processing ${files.length} Excel files...\n`);
insertAll(files);
db.exec("VACUUM");
const count = db.prepare("SELECT COUNT(*) as cnt FROM student").get();
console.log(`\nDone! ${count.cnt} students in database.`);
console.log(`Errors skipped: ${errorCount}`);
console.log(`Output: ${DB_PATH}`);
const stat = fs.statSync(DB_PATH);
console.log(`Size: ${(stat.size / 1024 / 1024).toFixed(1)} MB`);
db.close();
}
main();
+44
View File
@@ -0,0 +1,44 @@
#!/usr/bin/env bash
# Re-sync xlsxread Rust source from thptqg2017.
#
# Source SHA: 8b4a755c115595bf1b937d749eb1133efb3a6e22 (chore/xlsxread-rust)
#
# Re-run this when xlsxread is updated upstream. The configs/ directory is
# NOT synced — it contains thptqg2016-specific configs and test stubs that
# must be maintained here independently.
#
# Usage:
# ./tools/sync-from-thptqg2017.sh /path/to/thptqg2017
#
set -euo pipefail
SRC="${1:?Usage: $0 /path/to/thptqg2017}"
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
DEST="$SCRIPT_DIR/xlsxread"
if [ ! -d "$SRC/tools/xlsxread" ]; then
echo "Error: $SRC/tools/xlsxread not found" >&2
exit 1
fi
echo "Syncing from: $SRC/tools/xlsxread"
echo "Syncing to: $DEST"
echo ""
# Sync source, tests, and Cargo manifests — exclude build artifacts and dataset configs
for item in src tests Cargo.toml Cargo.lock; do
if [ -e "$SRC/tools/xlsxread/$item" ]; then
cp -r "$SRC/tools/xlsxread/$item" "$DEST/"
echo " synced: $item"
fi
done
echo ""
echo "Sync complete."
echo "Next steps:"
echo " 1. Review src/ for breaking changes to config.rs / writer.rs that"
echo " may affect format_detect_2016.rs or the thptqg2016-data.toml config."
echo " 2. Rebuild: cargo build --release --manifest-path $DEST/Cargo.toml"
echo " 3. Test: cargo test --manifest-path $DEST/Cargo.toml"
echo " 4. Commit on a chore/xlsxread-sync-... branch."
+1159
View File
File diff suppressed because it is too large. Load diff
+30
View File
@@ -0,0 +1,30 @@
[package]
name = "xlsxread"
version = "0.1.0"
edition = "2021"
description = "Rust CLI replacing SheetJS xlsx build scripts for thptqg2017/thptqg2016"
[dependencies]
calamine = "0.26"
rusqlite = { version = "0.32", features = ["bundled"] }
clap = { version = "4", features = ["derive"] }
serde = { version = "1", features = ["derive"] }
toml = "0.8"
regex = "1"
unicode-normalization = "0.1"
thiserror = "1"
anyhow = "1"
glob = "0.3"
# zip is already a transitive dep of calamine; pin explicitly so tests can use it
[dev-dependencies]
zip = "2"
rusqlite = { version = "0.32", features = ["bundled"] }
[[bin]]
name = "xlsxread"
path = "src/main.rs"
[[test]]
name = "golden"
path = "tests/golden.rs"
@@ -0,0 +1,86 @@
# Config for thptqg2016 data/ — 4 .xls + 115 .xlsx mixed files.
#
# Three column layouts exist across the 119 files; the binary selects the
# right one per-file at runtime via format_detection = "thptqg2016":
#
# separate-scores header SBD(0)/HOTEN(1)/TOAN(2)... — dhhanghai files
# mapped header SOBAODANH|SBD + DIEM_THI — most provinces
# default no header; positional 6-col layout — remaining files
#
# Schema differences from thptqg2017:
# + ten_cum_thi TEXT (exam-cluster name, TEN_CUMTHI column)
# + gioi_tinh TEXT (gender: "Nam"/"Nữ", GIOI_TINH column)
# + tieng_duc REAL (German)
# + tieng_nhat REAL (Japanese)
# - khtn, khxh, tieng_nga (not in 2016 dataset)
#
# sheet_mode = "all": several provinces overflow into Sheet2 (65k Excel row cap).
# strip_blank_rows = false: no blank-row anomaly observed in this dataset.
format_detection = "thptqg2016"
[reader]
sheet_mode = "all"
strip_blank_rows = false
[validation]
require_numeric_sbd = false
require_nonempty_name = true
require_nonempty_sbd = true
[header]
# Tokens that identify a header row by first-cell content (uppercased).
# Covers both SOBAODANH-style and SBD-style headers.
tokens = ["SOBAODANH", "SBD", "HO_TEN", "HOTEN", "HỌ TÊN", "STT"]
[schema]
ddl = """
CREATE TABLE student (
so_bao_danh TEXT PRIMARY KEY,
ho_ten TEXT NOT NULL,
ho_ten_ascii TEXT NOT NULL,
ngay_sinh TEXT,
ten_cum_thi TEXT,
gioi_tinh TEXT,
toan REAL,
ngu_van REAL,
vat_ly REAL,
hoa_hoc REAL,
sinh_hoc REAL,
lich_su REAL,
dia_ly REAL,
tieng_anh REAL,
tieng_phap REAL,
tieng_duc REAL,
tieng_nhat REAL,
tieng_trung REAL
);
CREATE INDEX idx_ho_ten ON student(ho_ten);
CREATE INDEX idx_ho_ten_ascii ON student(ho_ten_ascii);
CREATE INDEX idx_ten_cum_thi ON student(ten_cum_thi);
"""
[scores]
toan = 'Toán:\s*(\d+(?:\.\d+)?)'
ngu_van = 'Ngữ văn:\s*(\d+(?:\.\d+)?)'
vat_ly = 'Vật lí:\s*(\d+(?:\.\d+)?)'
hoa_hoc = 'Hóa học:\s*(\d+(?:\.\d+)?)'
sinh_hoc = 'Sinh học:\s*(\d+(?:\.\d+)?)'
lich_su = 'Lịch sử:\s*(\d+(?:\.\d+)?)'
dia_ly = 'Địa lí:\s*(\d+(?:\.\d+)?)'
tieng_anh = 'Tiếng Anh:\s*(\d+(?:\.\d+)?)'
tieng_phap = 'Tiếng Pháp:\s*(\d+(?:\.\d+)?)'
tieng_duc = 'Tiếng Đức:\s*(\d+(?:\.\d+)?)'
tieng_nhat = 'Tiếng Nhật:\s*(\d+(?:\.\d+)?)'
tieng_trung = 'Tiếng Trung:\s*(\d+(?:\.\d+)?)'
[insert]
sql = """
INSERT OR REPLACE INTO student
(so_bao_danh, ho_ten, ho_ten_ascii, ngay_sinh, ten_cum_thi, gioi_tinh,
toan, ngu_van, vat_ly, hoa_hoc, sinh_hoc,
lich_su, dia_ly,
tieng_anh, tieng_phap, tieng_duc, tieng_nhat, tieng_trung)
VALUES
(?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
"""
@@ -0,0 +1,76 @@
# Test-only config used by golden integration tests (data-old variant).
# Same column layout as thptqg2017-data.toml but with:
# - sheet_mode = "first" (reads only first sheet)
# - require_numeric_sbd = true (rejects non-digit SBDs)
# Schema and INSERT match SCORE_FIELDS_2017 (14 score cols).
[reader]
sheet_mode = "first"
strip_blank_rows = false
[columns]
ho_ten = 0
ngay_sinh = 1
so_bao_danh = 2
diem_thi = 3
[validation]
require_numeric_sbd = true
require_nonempty_name = true
require_nonempty_sbd = true
[header]
tokens = ["HO_TEN", "HỌ TÊN", "STT"]
[schema]
ddl = """
CREATE TABLE student (
so_bao_danh TEXT PRIMARY KEY,
ho_ten TEXT NOT NULL,
ho_ten_ascii TEXT NOT NULL,
ngay_sinh TEXT,
toan REAL,
ngu_van REAL,
vat_ly REAL,
hoa_hoc REAL,
sinh_hoc REAL,
khtn REAL,
lich_su REAL,
dia_ly REAL,
gdcd REAL,
khxh REAL,
tieng_anh REAL,
tieng_phap REAL,
tieng_nga REAL,
tieng_trung REAL
);
CREATE INDEX idx_ho_ten ON student(ho_ten);
CREATE INDEX idx_ho_ten_ascii ON student(ho_ten_ascii);
"""
[scores]
toan = 'Toán:\s*(\d+(?:\.\d+)?)'
ngu_van = 'Ngữ văn:\s*(\d+(?:\.\d+)?)'
vat_ly = 'Vật lí:\s*(\d+(?:\.\d+)?)'
hoa_hoc = 'Hóa học:\s*(\d+(?:\.\d+)?)'
sinh_hoc = 'Sinh học:\s*(\d+(?:\.\d+)?)'
khtn = 'KHTN:\s*(\d+(?:\.\d+)?)'
lich_su = 'Lịch sử:\s*(\d+(?:\.\d+)?)'
dia_ly = 'Địa lí:\s*(\d+(?:\.\d+)?)'
gdcd = 'GDCD:\s*(\d+(?:\.\d+)?)'
khxh = 'KHXH:\s*(\d+(?:\.\d+)?)'
tieng_anh = 'Tiếng Anh:\s*(\d+(?:\.\d+)?)'
tieng_phap = 'Tiếng Pháp:\s*(\d+(?:\.\d+)?)'
tieng_nga = 'Tiếng Nga:\s*(\d+(?:\.\d+)?)'
tieng_trung = 'Tiếng Trung:\s*(\d+(?:\.\d+)?)'
[insert]
sql = """
INSERT OR REPLACE INTO student
(so_bao_danh, ho_ten, ho_ten_ascii, ngay_sinh,
toan, ngu_van, vat_ly, hoa_hoc, sinh_hoc, khtn,
lich_su, dia_ly, gdcd, khxh,
tieng_anh, tieng_phap, tieng_nga, tieng_trung)
VALUES
(?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
"""
@@ -0,0 +1,76 @@
# Test-only config used by golden integration tests.
# Matches the fixture row layout produced by tests/golden.rs: write_xlsx()
# header: HO_TEN(0) NGAY_SINH(1) SO_BAO_DANH(2) DIEM_THI(3)
# Schema and INSERT match SCORE_FIELDS_2017 (14 score cols) so run_build_cmd
# in golden.rs can call insert_row(..., SCORE_FIELDS) without modification.
[reader]
sheet_mode = "all"
strip_blank_rows = false
[columns]
ho_ten = 0
ngay_sinh = 1
so_bao_danh = 2
diem_thi = 3
[validation]
require_numeric_sbd = false
require_nonempty_name = true
require_nonempty_sbd = true
[header]
tokens = ["HO_TEN", "HỌ TÊN", "STT"]
[schema]
ddl = """
CREATE TABLE student (
so_bao_danh TEXT PRIMARY KEY,
ho_ten TEXT NOT NULL,
ho_ten_ascii TEXT NOT NULL,
ngay_sinh TEXT,
toan REAL,
ngu_van REAL,
vat_ly REAL,
hoa_hoc REAL,
sinh_hoc REAL,
khtn REAL,
lich_su REAL,
dia_ly REAL,
gdcd REAL,
khxh REAL,
tieng_anh REAL,
tieng_phap REAL,
tieng_nga REAL,
tieng_trung REAL
);
CREATE INDEX idx_ho_ten ON student(ho_ten);
CREATE INDEX idx_ho_ten_ascii ON student(ho_ten_ascii);
"""
[scores]
toan = 'Toán:\s*(\d+(?:\.\d+)?)'
ngu_van = 'Ngữ văn:\s*(\d+(?:\.\d+)?)'
vat_ly = 'Vật lí:\s*(\d+(?:\.\d+)?)'
hoa_hoc = 'Hóa học:\s*(\d+(?:\.\d+)?)'
sinh_hoc = 'Sinh học:\s*(\d+(?:\.\d+)?)'
khtn = 'KHTN:\s*(\d+(?:\.\d+)?)'
lich_su = 'Lịch sử:\s*(\d+(?:\.\d+)?)'
dia_ly = 'Địa lí:\s*(\d+(?:\.\d+)?)'
gdcd = 'GDCD:\s*(\d+(?:\.\d+)?)'
khxh = 'KHXH:\s*(\d+(?:\.\d+)?)'
tieng_anh = 'Tiếng Anh:\s*(\d+(?:\.\d+)?)'
tieng_phap = 'Tiếng Pháp:\s*(\d+(?:\.\d+)?)'
tieng_nga = 'Tiếng Nga:\s*(\d+(?:\.\d+)?)'
tieng_trung = 'Tiếng Trung:\s*(\d+(?:\.\d+)?)'
[insert]
sql = """
INSERT OR REPLACE INTO student
(so_bao_danh, ho_ten, ho_ten_ascii, ngay_sinh,
toan, ngu_van, vat_ly, hoa_hoc, sinh_hoc, khtn,
lich_su, dia_ly, gdcd, khxh,
tieng_anh, tieng_phap, tieng_nga, tieng_trung)
VALUES
(?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
"""
+182
View File
@@ -0,0 +1,182 @@
/// Audit subcommand: replicates audit-row-counts.js exactly.
///
/// Reads all .xlsx files from the input directory (sheet 0 only, matching the
/// JS script's behaviour at audit-row-counts.js:33), collects distinct SBDs
/// into a HashSet, then queries `SELECT COUNT(*) FROM student` from the DB.
/// Prints the same lines as audit-row-counts.js:54-62 and exits 0 on match,
/// 1 on mismatch.
use std::collections::HashSet;
use std::path::Path;
use calamine::{open_workbook_auto, Data, Reader};
use crate::config::DatasetConfig;
use crate::error::BuildError;
use crate::reader::is_header_row;
// ---------------------------------------------------------------------------
// Audit result
// ---------------------------------------------------------------------------
pub struct AuditResult {
pub total_data_rows: u64,
pub both_empty: u64,
pub empty_name: u64,
pub empty_sbd: u64,
pub distinct_sbds: usize,
pub db_count: i64,
pub matched: bool,
}
// ---------------------------------------------------------------------------
// Main audit logic
// ---------------------------------------------------------------------------
/// Collect distinct SBDs from all xlsx files in `input_dir`, query `db_path`,
/// print the audit report and return the result.
///
/// The JS script reads only sheet 0 for every file (audit-row-counts.js:33).
/// Unlike build-database.js, the audit script does NOT iterate all sheets.
pub fn run_audit(
input_dir: &Path,
db_path: &Path,
cfg: &DatasetConfig,
) -> Result<AuditResult, BuildError> {
// Collect .xlsx files (audit-row-counts.js only checks .xlsx — line 15)
let mut files: Vec<std::path::PathBuf> = std::fs::read_dir(input_dir)
.map_err(|e| BuildError::Io {
path: input_dir.display().to_string(),
source: e,
})?
.filter_map(|e| e.ok())
.map(|e| e.path())
.filter(|p| {
p.is_file()
&& p.extension()
.and_then(|e| e.to_str())
.map(|e| e.eq_ignore_ascii_case("xlsx"))
.unwrap_or(false)
})
.collect();
files.sort();
let mut all_sbd: HashSet<String> = HashSet::new();
let mut total_data_rows: u64 = 0;
let mut empty_name: u64 = 0;
let mut empty_sbd: u64 = 0;
let mut both_empty: u64 = 0;
for file in &files {
let path_str = file.display().to_string();
let mut workbook = open_workbook_auto(file).map_err(|e| BuildError::Calamine {
path: path_str.clone(),
source: e,
})?;
let sheet_names = workbook.sheet_names().to_vec();
if sheet_names.is_empty() {
continue;
}
// audit-row-counts.js reads only sheet 0 (line 33: wb.SheetNames[0])
let range =
workbook
.worksheet_range(&sheet_names[0])
.map_err(|e| BuildError::Calamine {
path: path_str.clone(),
source: e,
})?;
let mut first_row = true;
for raw in range.rows() {
let row: Vec<Data> = raw.to_vec();
// Skip header row on first row only
if first_row {
first_row = false;
if is_header_row(&row, &cfg.header) {
continue;
}
}
total_data_rows += 1;
// For format_detection configs the audit uses positional defaults (col 0 = SBD,
// col 1 = HO_TEN) because the audit is best-effort and mirrors the JS script
// which also uses a fixed column assumption (audit-row-counts.js:33–36).
let (ho_ten_col, sbd_col) = cfg
.columns
.as_ref()
.map(|c| (c.ho_ten, c.so_bao_danh))
.unwrap_or((1, 0));
let ho_ten = row
.get(ho_ten_col)
.map(|c| c.to_string().trim().to_owned())
.unwrap_or_default();
let sbd = row
.get(sbd_col)
.map(|c| c.to_string().trim().to_owned())
.unwrap_or_default();
if ho_ten.is_empty() && sbd.is_empty() {
both_empty += 1;
continue;
}
if ho_ten.is_empty() {
empty_name += 1;
}
if sbd.is_empty() {
empty_sbd += 1;
}
if !sbd.is_empty() {
all_sbd.insert(sbd);
}
}
}
// Query DB count
let conn =
rusqlite::Connection::open_with_flags(db_path, rusqlite::OpenFlags::SQLITE_OPEN_READ_ONLY)?;
let db_count: i64 = conn.query_row("SELECT COUNT(*) FROM student", [], |row| row.get(0))?;
let distinct_sbds = all_sbd.len();
let matched = distinct_sbds as i64 == db_count;
Ok(AuditResult {
total_data_rows,
both_empty,
empty_name,
empty_sbd,
distinct_sbds,
db_count,
matched,
})
}
// ---------------------------------------------------------------------------
// Print audit report — mirrors audit-row-counts.js:54-62 exactly
// ---------------------------------------------------------------------------
pub fn print_audit_report(r: &AuditResult) {
println!("=== Source vs DB ===");
println!(
"Source: total data rows across all files: {}",
r.total_data_rows
);
println!(
"Source: rows with empty name AND sbd (skipped): {}",
r.both_empty
);
println!("Source: rows with missing name only: {}", r.empty_name);
println!("Source: rows with missing sbd only: {}", r.empty_sbd);
println!("Source: distinct SBDs: {}", r.distinct_sbds);
println!("DB: row count: {}", r.db_count);
println!(
"Match: {}",
if r.matched {
"YES — all unique SBDs accounted for".to_string()
} else {
format!("NO — gap of {}", r.distinct_sbds as i64 - r.db_count)
}
);
}
+48
View File
@@ -0,0 +1,48 @@
/// CLI argument structs via clap derive.
use std::path::PathBuf;
use clap::{Parser, Subcommand};
#[derive(Parser)]
#[command(
name = "xlsxread",
version,
about = "Read .xls/.xlsx files and build SQLite databases for thptqg datasets"
)]
pub struct Cli {
#[command(subcommand)]
pub cmd: Cmd,
}
#[derive(Subcommand)]
pub enum Cmd {
/// Read input spreadsheets and write a SQLite database
Build {
/// Path to the dataset TOML config file
#[arg(long)]
schema: PathBuf,
/// Directory containing the .xls / .xlsx source files
#[arg(long)]
input: PathBuf,
/// Output SQLite database path
#[arg(long)]
output: PathBuf,
},
/// Audit: compare distinct SBD count from xlsx files vs DB row count
Audit {
/// Path to the dataset TOML config file
#[arg(long)]
schema: PathBuf,
/// Directory containing the .xlsx source files
#[arg(long)]
input: PathBuf,
/// SQLite database to compare against
#[arg(long)]
db: PathBuf,
},
}
+188
View File
@@ -0,0 +1,188 @@
use std::collections::HashMap;
use std::fs;
use std::path::Path;
use serde::Deserialize;
use crate::error::BuildError;
// ---------------------------------------------------------------------------
// Top-level dataset configuration loaded from a .toml file
// ---------------------------------------------------------------------------
#[derive(Debug, Deserialize, Clone)]
pub struct DatasetConfig {
pub reader: ReaderCfg,
/// Fixed column indices. Optional when format_detection handles per-file mapping.
#[serde(default)]
pub columns: Option<ColumnMap>,
pub validation: ValidationCfg,
pub header: HeaderCfg,
pub schema: SchemaCfg,
/// field name → regex source string (one entry per scoreable subject)
pub scores: HashMap<String, String>,
pub insert: InsertCfg,
/// When set to "thptqg2016", enables per-file format auto-detection.
/// Each file's header row is inspected at runtime to choose the right
/// column layout (separate-scores / mapped / default-positional).
#[serde(default)]
pub format_detection: Option<String>,
}
#[derive(Debug, Deserialize, Clone)]
pub struct ReaderCfg {
/// "all" → iterate every sheet (handles HCM/HN overflow); "first" → sheet 0 only
pub sheet_mode: SheetMode,
/// If true, skip rows where every cell is empty/null before counting (data-old2 quirk)
pub strip_blank_rows: bool,
}
#[derive(Debug, Deserialize, Clone, PartialEq, Eq)]
#[serde(rename_all = "lowercase")]
pub enum SheetMode {
All,
First,
}
/// Zero-indexed column positions in the source spreadsheet row.
/// Used by thptqg2017 configs. thptqg2016 uses runtime format detection instead.
#[derive(Debug, Deserialize, Clone)]
pub struct ColumnMap {
pub ho_ten: usize,
pub ngay_sinh: usize,
pub so_bao_danh: usize,
pub diem_thi: usize,
}
#[derive(Debug, Deserialize, Clone)]
pub struct ValidationCfg {
/// build-database-old.js / -old2.js require soBaoDanh to match ^\d+$
pub require_numeric_sbd: bool,
pub require_nonempty_name: bool,
pub require_nonempty_sbd: bool,
}
#[derive(Debug, Deserialize, Clone)]
pub struct HeaderCfg {
/// Tokens to match against row[0].to_uppercase() to detect a header row
pub tokens: Vec<String>,
}
#[derive(Debug, Deserialize, Clone)]
pub struct SchemaCfg {
/// DDL executed verbatim before inserts (CREATE TABLE + CREATE INDEX)
pub ddl: String,
}
#[derive(Debug, Deserialize, Clone)]
pub struct InsertCfg {
/// Parameterised INSERT OR REPLACE SQL using :named_param style
pub sql: String,
}
// ---------------------------------------------------------------------------
// Loader
// ---------------------------------------------------------------------------
pub fn load_config(path: &Path) -> Result<DatasetConfig, BuildError> {
let text = fs::read_to_string(path).map_err(|e| BuildError::Io {
path: path.display().to_string(),
source: e,
})?;
let cfg: DatasetConfig = toml::from_str(&text)?;
Ok(cfg)
}
// ---------------------------------------------------------------------------
// Unit tests
// ---------------------------------------------------------------------------
#[cfg(test)]
mod tests {
use super::*;
const SAMPLE_TOML: &str = r#"
[reader]
sheet_mode = "all"
strip_blank_rows = false
[columns]
ho_ten = 0
ngay_sinh = 1
so_bao_danh = 2
diem_thi = 3
[validation]
require_numeric_sbd = false
require_nonempty_name = true
require_nonempty_sbd = true
[header]
tokens = ["HO_TEN", "HỌ TÊN", "STT"]
[schema]
ddl = "CREATE TABLE student (so_bao_danh TEXT PRIMARY KEY);"
[scores]
toan = 'Toán:\s*(\d+(?:\.\d+)?)'
ngu_van = 'Ngữ văn:\s*(\d+(?:\.\d+)?)'
[insert]
sql = "INSERT OR REPLACE INTO student (so_bao_danh) VALUES (:so_bao_danh)"
"#;
#[test]
fn config_round_trip() {
let cfg: DatasetConfig = toml::from_str(SAMPLE_TOML).expect("parse failed");
assert_eq!(cfg.reader.sheet_mode, SheetMode::All);
assert!(!cfg.reader.strip_blank_rows);
let cols = cfg.columns.as_ref().unwrap();
assert_eq!(cols.ho_ten, 0);
assert_eq!(cols.diem_thi, 3);
assert!(!cfg.validation.require_numeric_sbd);
assert!(cfg.validation.require_nonempty_name);
assert_eq!(cfg.header.tokens.len(), 3);
assert!(cfg.scores.contains_key("toan"));
assert!(cfg.scores.contains_key("ngu_van"));
assert!(cfg.format_detection.is_none());
}
#[test]
fn config_first_sheet_mode() {
let toml_str = SAMPLE_TOML.replace(r#"sheet_mode = "all""#, r#"sheet_mode = "first""#);
let cfg: DatasetConfig = toml::from_str(&toml_str).expect("parse failed");
assert_eq!(cfg.reader.sheet_mode, SheetMode::First);
}
#[test]
fn config_format_detection_field() {
// Configs without [columns] and with format_detection = "thptqg2016" parse correctly
let toml_str = r#"
format_detection = "thptqg2016"
[reader]
sheet_mode = "all"
strip_blank_rows = false
[validation]
require_numeric_sbd = false
require_nonempty_name = true
require_nonempty_sbd = true
[header]
tokens = ["SBD", "SOBAODANH", "STT"]
[schema]
ddl = "CREATE TABLE student (so_bao_danh TEXT PRIMARY KEY);"
[scores]
toan = 'Toán:\s*(\d+(?:\.\d+)?)'
[insert]
sql = "INSERT OR REPLACE INTO student (so_bao_danh) VALUES (?)"
"#;
let cfg: DatasetConfig = toml::from_str(toml_str).expect("parse failed");
assert_eq!(cfg.format_detection.as_deref(), Some("thptqg2016"));
assert!(cfg.columns.is_none());
}
}
+34
View File
@@ -0,0 +1,34 @@
use thiserror::Error;
#[derive(Debug, Error)]
pub enum BuildError {
#[error("I/O error for {path}: {source}")]
Io {
path: String,
#[source]
source: std::io::Error,
},
#[error("Calamine error for {path}: {source}")]
Calamine {
path: String,
#[source]
source: calamine::Error,
},
#[error("SQLite error: {0}")]
Sqlite(#[from] rusqlite::Error),
#[error("Config parse error: {0}")]
Config(#[from] toml::de::Error),
#[error("Regex compile error for pattern '{pattern}': {source}")]
Regex {
pattern: String,
#[source]
source: regex::Error,
},
#[error("Schema has no sheets in file: {0}")]
NoSheets(String),
}
@@ -0,0 +1,549 @@
/// Per-file format auto-detection for the thptqg2016 dataset.
///
/// Translates `detectFormat` from scripts/build-database.js (lines 63–87) and the
/// three row-processing functions (lines 90–146) into Rust.
///
/// The JS source has three formats:
///
/// 1. `separate-scores` — header row[0]=="SBD" && row[2]=="TOAN"
/// Columns: SBD(0) HOTEN(1) TOAN(2) VAN(3) LY(4) HOA(5) SINH(6) SU(7) DIA(8)
/// NGOAINGUTN(9) NGOAINGUTL(10) NGOAINGU-total(11)
/// → maps col 11 → tieng_anh; no ngay_sinh / ten_cum_thi / gioi_tinh / DIEM_THI
/// → JS: build-database.js:90–116 (processSeparateScoresRow)
///
/// 2. `mapped` — header row has SOBAODANH|SBD and DIEM_THI columns
/// → dynamic column indices built from header names
/// → JS: build-database.js:119–146 (processMappedRow with map from detectFormat)
///
/// 3. `default` — no recognised header; positional 6-col layout
/// SBD(0) HO_TEN(1) NGAY_SINH(2) TEN_CUMTHI(3) GIOI_TINH(4) DIEM_THI(5)
/// → JS: build-database.js:149–151 DEFAULT_MAP + processMappedRow
///
/// JS citations are line numbers in /config/workspace/tiennm99/thptqg2016/scripts/build-database.js.
use calamine::Data;
use crate::transform::{parse_scores, to_ascii, CompiledPatterns, ParsedRow};
// ---------------------------------------------------------------------------
// Known header tokens (mirrors JS KNOWN_HEADERS set, build-database.js:50–54)
// ---------------------------------------------------------------------------
/// Upper-cased strings that identify a header row's first cell.
/// Mirrors `KNOWN_HEADERS` in build-database.js:50–54.
const KNOWN_HEADERS: &[&str] = &[
"SOBAODANH",
"SBD",
"HO_TEN",
"HOTEN",
"HỌ TÊN",
"NGAY_SINH",
"TEN_CUMTHI",
"GIOI_TINH",
"DIEM_THI",
"STT",
"TOAN",
"VAN",
"LY",
"HOA",
"SINH ",
"SU",
"DIA",
];
/// Returns true when `row[0]` (uppercased, trimmed) is in the known-headers set.
/// Mirrors `isHeaderRow` at build-database.js:56–60.
pub fn is_header_row_2016(row: &[Data]) -> bool {
if row.len() < 2 {
return false;
}
let first = row[0].to_string().trim().to_uppercase();
KNOWN_HEADERS.iter().any(|h| *h == first.as_str())
}
// ---------------------------------------------------------------------------
// Detected per-file format
// ---------------------------------------------------------------------------
/// The three layouts the thptqg2016 dataset uses, detected per file.
#[derive(Debug, Clone)]
pub enum DetectedFormat {
/// SBD/HOTEN/TOAN/VAN/LY/HOA/SINH/SU/DIA/NGOAINGUTN/NGOAINGUTL/NGOAINGU columns.
/// Corresponds to the dhhanghai-style files. build-database.js:67–68.
SeparateScores,
/// Header present with SOBAODANH|SBD and DIEM_THI; dynamic column indices.
/// build-database.js:70–86.
Mapped {
sbd: usize,
ho_ten: usize,
ngay_sinh: Option<usize>,
ten_cum_thi: Option<usize>,
gioi_tinh: Option<usize>,
diem_thi: usize,
},
/// No recognised header; standard 6-column positional layout.
/// build-database.js:149–151 DEFAULT_MAP.
Default,
}
/// Inspect a header row and decide which format applies.
/// Returns `None` when the header is present but unrecognised (treated as Default).
///
/// Mirrors `detectFormat` at build-database.js:63–87.
pub fn detect_format(header_row: &[Data]) -> DetectedFormat {
let cols: Vec<String> = header_row
.iter()
.map(|c| c.to_string().trim().to_uppercase())
.collect();
// Format 1: SBD in col 0 AND TOAN in col 2 → separate-scores
// build-database.js:68: if (cols[0] === "SBD" && cols[2] === "TOAN")
if cols.first().map(|s| s.as_str()) == Some("SBD")
&& cols.get(2).map(|s| s.as_str()) == Some("TOAN")
{
return DetectedFormat::SeparateScores;
}
// Format 2: build column index map — check for SOBAODANH|SBD and DIEM_THI
// build-database.js:70–86
let mut sbd_idx: Option<usize> = None;
let mut ho_ten_idx: Option<usize> = None;
let mut ngay_sinh_idx: Option<usize> = None;
let mut ten_cum_thi_idx: Option<usize> = None;
let mut gioi_tinh_idx: Option<usize> = None;
let mut diem_thi_idx: Option<usize> = None;
for (i, c) in cols.iter().enumerate() {
match c.as_str() {
"SOBAODANH" | "SBD" => sbd_idx = Some(i),
"HO_TEN" | "HOTEN" | "HỌ TÊN" => ho_ten_idx = Some(i),
"NGAY_SINH" => ngay_sinh_idx = Some(i),
"TEN_CUMTHI" => ten_cum_thi_idx = Some(i),
"GIOI_TINH" => gioi_tinh_idx = Some(i),
"DIEM_THI" => diem_thi_idx = Some(i),
_ => {}
}
}
// build-database.js:82–84: if (map.sbd !== undefined && map.diem_thi !== undefined)
if let (Some(sbd), Some(diem_thi)) = (sbd_idx, diem_thi_idx) {
let ho_ten = ho_ten_idx.unwrap_or(1); // fallback: col 1 (present in all known files)
return DetectedFormat::Mapped {
sbd,
ho_ten,
ngay_sinh: ngay_sinh_idx,
ten_cum_thi: ten_cum_thi_idx,
gioi_tinh: gioi_tinh_idx,
diem_thi,
};
}
// Unrecognised header (or no header) → positional default
DetectedFormat::Default
}
// ---------------------------------------------------------------------------
// Row processors
// ---------------------------------------------------------------------------
/// Cell accessor helper.
fn cell_str(row: &[Data], idx: usize) -> String {
row.get(idx)
.map(|c| c.to_string().trim().to_owned())
.unwrap_or_default()
}
/// Parse a cell that should hold a float score; returns None for blank/non-numeric.
/// Mirrors `parseFloat(row[N]) || null` in JS.
fn parse_float_cell(row: &[Data], idx: usize) -> Option<f64> {
let s = cell_str(row, idx);
if s.is_empty() {
return None;
}
s.parse::<f64>().ok().filter(|v| v.is_finite() && *v != 0.0)
}
/// Process a row in the `separate-scores` format.
///
/// Column layout (build-database.js:90–116 `processSeparateScoresRow`):
/// 0=SBD 1=HOTEN 2=TOAN 3=VAN 4=LY 5=HOA 6=SINH 7=SU 8=DIA
/// 9=NGOAINGUTN 10=NGOAINGUTL 11=NGOAINGU(total→tieng_anh)
///
/// tieng_phap / tieng_duc / tieng_nhat / tieng_trung all → None
/// ngay_sinh / ten_cum_thi / gioi_tinh all → None (not in this format)
pub fn process_separate_scores_row(row: &[Data], patterns: &CompiledPatterns) -> Option<ParsedRow> {
let sbd = cell_str(row, 0);
let ho_ten = cell_str(row, 1);
if sbd.is_empty() || ho_ten.is_empty() {
return None;
}
let ho_ten_ascii = to_ascii(&ho_ten);
// build-database.js:102–114: explicit per-column score mapping
let mut scores = std::collections::HashMap::new();
macro_rules! add_score {
($field:expr, $idx:expr) => {
if let Some(v) = parse_float_cell(row, $idx) {
scores.insert($field.to_string(), v);
}
};
}
add_score!("toan", 2);
add_score!("ngu_van", 3);
add_score!("vat_ly", 4);
add_score!("hoa_hoc", 5);
add_score!("sinh_hoc", 6);
add_score!("lich_su", 7);
add_score!("dia_ly", 8);
// col 11 = NGOAINGU total → tieng_anh (build-database.js:110–111)
add_score!("tieng_anh", 11);
// Suppress unused-variable warning; patterns not used in this path (no DIEM_THI string)
let _ = patterns;
Some(ParsedRow {
so_bao_danh: sbd,
ho_ten,
ho_ten_ascii,
ngay_sinh: None,
ten_cum_thi: None,
gioi_tinh: None,
scores,
})
}
/// Process a row in the `mapped` format (header-derived column indices).
///
/// Mirrors `processMappedRow` at build-database.js:119–146.
/// Gender is normalised: only "Nam" or "Nữ" are kept; everything else → None.
/// (build-database.js:132: `(rawGioiTinh === "Nam" || rawGioiTinh === "Nữ") ? rawGioiTinh : null`)
pub fn process_mapped_row(
row: &[Data],
sbd_idx: usize,
ho_ten_idx: usize,
ngay_sinh_idx: Option<usize>,
ten_cum_thi_idx: Option<usize>,
gioi_tinh_idx: Option<usize>,
diem_thi_idx: usize,
patterns: &CompiledPatterns,
) -> Option<ParsedRow> {
let sbd = cell_str(row, sbd_idx);
let ho_ten = cell_str(row, ho_ten_idx);
if sbd.is_empty() || ho_ten.is_empty() {
return None;
}
// Skip rows where SBD or HO_TEN are themselves header tokens (leaked header rows).
// build-database.js:125–126: KNOWN_HEADERS.has(sbdUpper) || KNOWN_HEADERS.has(hoTenUpper)
let sbd_upper = sbd.to_uppercase();
let ho_ten_upper = ho_ten.to_uppercase();
if KNOWN_HEADERS.iter().any(|h| *h == sbd_upper.as_str())
|| KNOWN_HEADERS.iter().any(|h| *h == ho_ten_upper.as_str())
{
return None;
}
let ho_ten_ascii = to_ascii(&ho_ten);
let ngay_sinh = ngay_sinh_idx
.map(|i| cell_str(row, i))
.filter(|s| !s.is_empty());
let ten_cum_thi = ten_cum_thi_idx
.map(|i| cell_str(row, i))
.filter(|s| !s.is_empty());
// build-database.js:130–132: normalise gender
let gioi_tinh = gioi_tinh_idx
.map(|i| cell_str(row, i))
.and_then(|s| {
if s == "Nam" || s == "Nữ" {
Some(s)
} else {
None
}
});
let diem_thi = row
.get(diem_thi_idx)
.map(|c| c.to_string())
.unwrap_or_default();
let scores = parse_scores(&diem_thi, patterns);
Some(ParsedRow {
so_bao_danh: sbd,
ho_ten,
ho_ten_ascii,
ngay_sinh,
ten_cum_thi,
gioi_tinh,
scores,
})
}
/// Process a row using the default positional 6-column layout.
///
/// Column order: SBD(0) HO_TEN(1) NGAY_SINH(2) TEN_CUMTHI(3) GIOI_TINH(4) DIEM_THI(5)
/// Mirrors `processMappedRow(row, DEFAULT_MAP)` at build-database.js:233–236.
pub fn process_default_row(row: &[Data], patterns: &CompiledPatterns) -> Option<ParsedRow> {
process_mapped_row(
row,
0, // sbd
1, // ho_ten
Some(2),
Some(3),
Some(4),
5, // diem_thi
patterns,
)
}
/// Dispatch a data row through the correct processor for the detected format.
///
/// Returns `None` when the row is empty/invalid and should be skipped.
pub fn process_row_2016(
row: &[Data],
fmt: &DetectedFormat,
patterns: &CompiledPatterns,
) -> Option<ParsedRow> {
match fmt {
DetectedFormat::SeparateScores => process_separate_scores_row(row, patterns),
DetectedFormat::Mapped {
sbd,
ho_ten,
ngay_sinh,
ten_cum_thi,
gioi_tinh,
diem_thi,
} => process_mapped_row(
row,
*sbd,
*ho_ten,
*ngay_sinh,
*ten_cum_thi,
*gioi_tinh,
*diem_thi,
patterns,
),
DetectedFormat::Default => process_default_row(row, patterns),
}
}
// ---------------------------------------------------------------------------
// Unit tests — 3 detection branches + key processing cases
// ---------------------------------------------------------------------------
#[cfg(test)]
mod tests {
use super::*;
use std::collections::HashMap;
fn s(v: &str) -> Data {
Data::String(v.to_string())
}
fn make_patterns() -> CompiledPatterns {
let mut map = HashMap::new();
map.insert("toan".into(), r"Toán:\s*(\d+(?:\.\d+)?)".into());
map.insert("ngu_van".into(), r"Ngữ văn:\s*(\d+(?:\.\d+)?)".into());
map.insert("tieng_anh".into(), r"Tiếng Anh:\s*(\d+(?:\.\d+)?)".into());
map.insert("tieng_duc".into(), r"Tiếng Đức:\s*(\d+(?:\.\d+)?)".into());
map.insert("tieng_nhat".into(), r"Tiếng Nhật:\s*(\d+(?:\.\d+)?)".into());
CompiledPatterns::new(&map).unwrap()
}
// --- detect_format: branch 1 — separate-scores ---
#[test]
fn detect_separate_scores() {
let header = vec![s("SBD"), s("HOTEN"), s("TOAN"), s("VAN")];
match detect_format(&header) {
DetectedFormat::SeparateScores => {}
other => panic!("expected SeparateScores, got {other:?}"),
}
}
// --- detect_format: branch 2 — mapped ---
#[test]
fn detect_mapped_with_named_cols() {
let header = vec![
s("SOBAODANH"),
s("HO_TEN"),
s("NGAY_SINH"),
s("TEN_CUMTHI"),
s("GIOI_TINH"),
s("DIEM_THI"),
];
match detect_format(&header) {
DetectedFormat::Mapped {
sbd,
ho_ten,
ngay_sinh,
ten_cum_thi,
gioi_tinh,
diem_thi,
} => {
assert_eq!(sbd, 0);
assert_eq!(ho_ten, 1);
assert_eq!(ngay_sinh, Some(2));
assert_eq!(ten_cum_thi, Some(3));
assert_eq!(gioi_tinh, Some(4));
assert_eq!(diem_thi, 5);
}
other => panic!("expected Mapped, got {other:?}"),
}
}
#[test]
fn detect_mapped_sbd_variant() {
// "SBD" (not "SOBAODANH") + DIEM_THI at different positions
let header = vec![s("STT"), s("SBD"), s("HOTEN"), s("DIEM_THI")];
match detect_format(&header) {
DetectedFormat::Mapped { sbd, diem_thi, .. } => {
assert_eq!(sbd, 1);
assert_eq!(diem_thi, 3);
}
other => panic!("expected Mapped, got {other:?}"),
}
}
// --- detect_format: branch 3 — default ---
#[test]
fn detect_default_when_no_header() {
// A data row: positional default used
let data_row = vec![
s("12345678"),
s("Nguyễn Văn A"),
s("01/01/2000"),
s("TP HCM"),
s("Nam"),
s("Toán: 8.5"),
];
// Default is returned when there is no recognised header
match detect_format(&data_row) {
DetectedFormat::Default => {}
other => panic!("expected Default, got {other:?}"),
}
}
// --- process_separate_scores_row ---
#[test]
fn separate_scores_basic() {
let p = make_patterns();
// 12 columns: SBD HOTEN TOAN VAN LY HOA SINH SU DIA NGUTN NGUTL NGUTOTAL
let row = vec![
s("TP001"),
s("Nguyễn Thị Lan"),
Data::Float(8.0),
Data::Float(7.5),
Data::Float(9.0),
Data::Float(6.5),
Data::Float(5.0),
Data::Float(4.5),
Data::Float(8.0),
Data::Empty,
Data::Empty,
Data::Float(7.0), // col 11 → tieng_anh
];
let row = process_separate_scores_row(&row, &p).expect("should parse");
assert_eq!(row.so_bao_danh, "TP001");
assert_eq!(row.ho_ten, "Nguyễn Thị Lan");
assert_eq!(row.ho_ten_ascii, "nguyen thi lan");
assert_eq!(row.scores.get("toan"), Some(&8.0));
assert_eq!(row.scores.get("tieng_anh"), Some(&7.0));
assert!(row.ngay_sinh.is_none());
assert!(row.ten_cum_thi.is_none());
assert!(row.gioi_tinh.is_none());
}
#[test]
fn separate_scores_skips_empty_sbd() {
let p = make_patterns();
let row = vec![s(""), s("Nguyễn Văn A"), Data::Float(5.0)];
assert!(process_separate_scores_row(&row, &p).is_none());
}
// --- process_mapped_row ---
#[test]
fn mapped_row_full_fields() {
let p = make_patterns();
let row = vec![
s("HCM001"),
s("Trần Thị Bích"),
s("15/3/1999"),
s("Cụm thi HCM"),
s("Nữ"),
s("Toán: 9.0 Ngữ văn: 8.5 Tiếng Anh: 7.75"),
];
let parsed = process_mapped_row(&row, 0, 1, Some(2), Some(3), Some(4), 5, &p)
.expect("should parse");
assert_eq!(parsed.so_bao_danh, "HCM001");
assert_eq!(parsed.ngay_sinh.as_deref(), Some("15/3/1999"));
assert_eq!(parsed.ten_cum_thi.as_deref(), Some("Cụm thi HCM"));
assert_eq!(parsed.gioi_tinh.as_deref(), Some("Nữ"));
assert_eq!(parsed.scores.get("toan"), Some(&9.0));
assert_eq!(parsed.scores.get("ngu_van"), Some(&8.5));
assert_eq!(parsed.scores.get("tieng_anh"), Some(&7.75));
}
#[test]
fn mapped_row_gender_normalisation() {
let p = make_patterns();
// Gender "Unknown" → None
let row = vec![s("ABC"), s("Nguyen Van A"), s(""), s(""), s("Unknown"), s("")];
let parsed =
process_mapped_row(&row, 0, 1, Some(2), Some(3), Some(4), 5, &p).expect("should parse");
assert!(parsed.gioi_tinh.is_none());
// "Nam" passes through
let row2 = vec![s("ABC"), s("Nguyen Van A"), s(""), s(""), s("Nam"), s("")];
let parsed2 =
process_mapped_row(&row2, 0, 1, Some(2), Some(3), Some(4), 5, &p).expect("should parse");
assert_eq!(parsed2.gioi_tinh.as_deref(), Some("Nam"));
}
#[test]
fn mapped_row_skips_leaked_header() {
let p = make_patterns();
// A leaked header row — SBD cell contains "SOBAODANH"
let row = vec![s("SOBAODANH"), s("HO_TEN"), s("NGAY_SINH"), s(""), s(""), s("")];
assert!(process_mapped_row(&row, 0, 1, Some(2), Some(3), Some(4), 5, &p).is_none());
}
// --- process_default_row ---
#[test]
fn default_row_positional() {
let p = make_patterns();
let row = vec![
s("DN001"),
s("Lê Văn Long"),
s("10/5/1998"),
s("Cụm Đà Nẵng"),
s("Nam"),
s("Tiếng Đức: 6.25"),
];
let parsed = process_default_row(&row, &p).expect("should parse");
assert_eq!(parsed.so_bao_danh, "DN001");
assert_eq!(parsed.ho_ten_ascii, "le van long");
assert_eq!(parsed.ngay_sinh.as_deref(), Some("10/5/1998"));
assert_eq!(parsed.ten_cum_thi.as_deref(), Some("Cụm Đà Nẵng"));
assert_eq!(parsed.gioi_tinh.as_deref(), Some("Nam"));
assert_eq!(parsed.scores.get("tieng_duc"), Some(&6.25));
}
// --- is_header_row_2016 ---
#[test]
fn header_row_detection_2016() {
assert!(is_header_row_2016(&[s("SOBAODANH"), s("HO_TEN")]));
assert!(is_header_row_2016(&[s("SBD"), s("HOTEN"), s("TOAN")]));
assert!(!is_header_row_2016(&[s("12345678"), s("Nguyen Van A")]));
assert!(!is_header_row_2016(&[s("SBD")])); // too short (< 2 cells)
}
}
+10
View File
@@ -0,0 +1,10 @@
/// Public library interface for integration tests.
/// The binary entry point is src/main.rs; this file re-exports the internal
/// modules so tests/golden.rs can call them without going through the CLI.
pub mod audit;
pub mod config;
pub mod error;
pub mod format_detect_2016;
pub mod reader;
pub mod transform;
pub mod writer;
+379
View File
@@ -0,0 +1,379 @@
/// xlsxread — Rust CLI replacing the SheetJS xlsx build scripts.
///
/// Subcommands:
/// build — read .xls/.xlsx files → write SQLite DB
/// audit — compare distinct SBD count from xlsx vs DB row count
///
/// Library modules are declared in lib.rs; main.rs only adds the CLI layer.
///
/// When config contains `format_detection = "thptqg2016"` the build subcommand
/// uses per-file header inspection to pick the right column layout, replicating
/// the `detectFormat` logic from scripts/build-database.js (lines 63–87).
mod cli;
use std::path::Path;
use anyhow::{Context, Result};
use calamine::Data;
use clap::Parser;
use cli::{Cli, Cmd};
use xlsxread::audit;
use xlsxread::config::load_config;
use xlsxread::format_detect_2016::{
detect_format, is_header_row_2016, process_row_2016, DetectedFormat,
};
use xlsxread::reader::{is_all_blank, process_file};
use xlsxread::transform::{validate_row, CompiledPatterns, SkipReason};
use xlsxread::writer::{
finish_db, insert_row, insert_row_2016, open_db, SCORE_FIELDS, SCORE_FIELDS_2016,
};
fn main() -> Result<()> {
let cli = Cli::parse();
match cli.cmd {
Cmd::Build {
schema,
input,
output,
} => {
run_build(&schema, &input, &output)?;
}
Cmd::Audit { schema, input, db } => {
let cfg = load_config(&schema)
.with_context(|| format!("Failed to load config: {}", schema.display()))?;
let result = audit::run_audit(&input, &db, &cfg).with_context(|| "Audit failed")?;
audit::print_audit_report(&result);
if !result.matched {
std::process::exit(1);
}
}
}
Ok(())
}
// ---------------------------------------------------------------------------
// Build subcommand — dispatches to thptqg2016 or standard path
// ---------------------------------------------------------------------------
fn run_build(schema_path: &Path, input_dir: &Path, output_path: &Path) -> Result<()> {
let cfg = load_config(schema_path)
.with_context(|| format!("Failed to load config: {}", schema_path.display()))?;
if cfg.format_detection.as_deref() == Some("thptqg2016") {
run_build_2016(&cfg, input_dir, output_path)
} else {
run_build_standard(&cfg, input_dir, output_path)
}
}
// ---------------------------------------------------------------------------
// Standard build path (thptqg2017 and similar fixed-column configs)
// ---------------------------------------------------------------------------
fn run_build_standard(
cfg: &xlsxread::config::DatasetConfig,
input_dir: &Path,
output_path: &Path,
) -> Result<()> {
let patterns =
CompiledPatterns::new(&cfg.scores).with_context(|| "Failed to compile score regexes")?;
let mut files: Vec<std::path::PathBuf> = std::fs::read_dir(input_dir)
.with_context(|| format!("Cannot read input dir: {}", input_dir.display()))?
.filter_map(|e| e.ok())
.map(|e| e.path())
.filter(|p| {
p.is_file()
&& p.extension()
.and_then(|e| e.to_str())
.map(|e| {
let lower = e.to_lowercase();
lower == "xls" || lower == "xlsx"
})
.unwrap_or(false)
})
.collect();
files.sort();
let dataset_label = input_dir
.file_name()
.and_then(|n| n.to_str())
.unwrap_or("data");
println!(
"[build] {dataset_label}/ → {} ({} files)",
output_path.display(),
files.len()
);
let conn = open_db(output_path, cfg)
.with_context(|| format!("Failed to open DB: {}", output_path.display()))?;
let mut total_source_rows: u64 = 0;
let mut total_skipped: u64 = 0;
let mut total_errors: u64 = 0;
let is_old2 = dataset_label.contains("old2");
let strip_blank = cfg.reader.strip_blank_rows;
conn.execute_batch("BEGIN")?;
for file in &files {
let base = file
.file_name()
.and_then(|n| n.to_str())
.unwrap_or("?")
.to_owned();
let mut file_rows: u64 = 0;
let mut file_skipped: u64 = 0;
let mut file_errors: u64 = 0;
let process_result = process_file(file, cfg, |_sheet_idx, raw| {
let all_blank = is_all_blank(raw);
if strip_blank && all_blank {
return;
}
total_source_rows += 1;
let ho_ten = raw
.get(cfg.columns.as_ref().unwrap().ho_ten)
.map(|c| c.to_string().trim().to_owned())
.unwrap_or_default();
let so_bao_danh = raw
.get(cfg.columns.as_ref().unwrap().so_bao_danh)
.map(|c| c.to_string().trim().to_owned())
.unwrap_or_default();
match validate_row(&ho_ten, &so_bao_danh, &cfg.validation, strip_blank, all_blank) {
Err(SkipReason::BlankRow) => {}
Err(_) => {
file_skipped += 1;
return;
}
Ok(()) => {}
}
let parsed = xlsxread::transform::transform_row(raw, cfg, &patterns);
match insert_row(&conn, &cfg.insert.sql, &parsed, SCORE_FIELDS) {
Ok(()) => file_rows += 1,
Err(e) => {
file_errors += 1;
if total_errors + file_errors <= 5 {
eprintln!(" [warn] {base}: {e}");
}
}
}
});
match process_result {
Ok(_) => {}
Err(e) => {
eprintln!(" [error] {base}: {e}");
file_errors += 1;
}
}
total_skipped += file_skipped;
total_errors += file_errors;
println!(" {base}: {file_rows} rows");
}
conn.execute_batch("COMMIT")?;
finish_db(
&conn,
output_path,
total_source_rows,
total_skipped,
total_errors,
dataset_label,
files.len(),
is_old2,
)
.with_context(|| "Failed to finalise DB")?;
Ok(())
}
// ---------------------------------------------------------------------------
// thptqg2016 build path — per-file format detection
// ---------------------------------------------------------------------------
/// Build the thptqg2016 database.
///
/// Each file is processed independently: the first row is inspected to determine
/// which of the three column layouts applies (separate-scores / mapped / default).
/// This mirrors `detectFormat` in scripts/build-database.js lines 63–87, called
/// once per file inside the file loop at build-database.js:218–219.
fn run_build_2016(
cfg: &xlsxread::config::DatasetConfig,
input_dir: &Path,
output_path: &Path,
) -> Result<()> {
let patterns =
CompiledPatterns::new(&cfg.scores).with_context(|| "Failed to compile score regexes")?;
let mut files: Vec<std::path::PathBuf> = std::fs::read_dir(input_dir)
.with_context(|| format!("Cannot read input dir: {}", input_dir.display()))?
.filter_map(|e| e.ok())
.map(|e| e.path())
.filter(|p| {
p.is_file()
&& p.extension()
.and_then(|e| e.to_str())
.map(|e| {
let lower = e.to_lowercase();
lower == "xls" || lower == "xlsx"
})
.unwrap_or(false)
})
.collect();
files.sort();
let dataset_label = input_dir
.file_name()
.and_then(|n| n.to_str())
.unwrap_or("data");
println!(
"[build:2016] {dataset_label}/ → {} ({} files)",
output_path.display(),
files.len()
);
let conn = open_db(output_path, cfg)
.with_context(|| format!("Failed to open DB: {}", output_path.display()))?;
let mut total_source_rows: u64 = 0;
let total_skipped: u64 = 0;
let mut total_errors: u64 = 0;
conn.execute_batch("BEGIN")?;
for file in &files {
let base = file
.file_name()
.and_then(|n| n.to_str())
.unwrap_or("?")
.to_owned();
match process_file_2016(
file,
cfg,
&patterns,
&conn,
&base,
&mut total_source_rows,
&mut total_errors,
) {
Ok(file_rows) => {
println!(" {base}: {file_rows} rows");
}
Err(e) => {
eprintln!(" [error] {base}: {e}");
total_errors += 1;
}
}
}
conn.execute_batch("COMMIT")?;
finish_db(
&conn,
output_path,
total_source_rows,
total_skipped,
total_errors,
dataset_label,
files.len(),
false,
)
.with_context(|| "Failed to finalise DB")?;
Ok(())
}
/// Process one file in the thptqg2016 format-detection path.
///
/// Reads the file, uses the first row to detect the column layout, then processes
/// all subsequent data rows. Returns the count of successfully inserted rows.
fn process_file_2016(
file: &Path,
cfg: &xlsxread::config::DatasetConfig,
patterns: &CompiledPatterns,
conn: &rusqlite::Connection,
base: &str,
total_source_rows: &mut u64,
total_errors: &mut u64,
) -> Result<u64> {
use calamine::{open_workbook_auto, Reader, Sheets};
let path_str = file.display().to_string();
let mut workbook: Sheets<_> =
open_workbook_auto(file).with_context(|| format!("Cannot open {path_str}"))?;
let sheet_names: Vec<String> = workbook.sheet_names().to_vec();
if sheet_names.is_empty() {
return Ok(0);
}
// Sheet selection: thptqg2016 data/ has all-sheets mode to handle
// HCM/HN overflow (same reason as thptqg2017 data/).
let sheets_to_read: Vec<String> = match cfg.reader.sheet_mode {
xlsxread::config::SheetMode::All => sheet_names.clone(),
xlsxread::config::SheetMode::First => vec![sheet_names[0].clone()],
};
let mut file_rows: u64 = 0;
for sheet_name in &sheets_to_read {
let range = workbook
.worksheet_range(sheet_name)
.with_context(|| format!("Cannot read sheet {sheet_name} in {path_str}"))?;
let rows: Vec<Vec<Data>> = range.rows().map(|r| r.to_vec()).collect();
if rows.is_empty() {
continue;
}
// Detect format from first row, then determine start index.
// Mirrors build-database.js:215–220: isHeaderRow check + detectFormat.
let (fmt, start_idx) = if is_header_row_2016(&rows[0]) {
(detect_format(&rows[0]), 1)
} else {
(DetectedFormat::Default, 0)
};
for row in rows.iter().skip(start_idx) {
if row.len() < 2 {
continue;
}
*total_source_rows += 1;
match process_row_2016(row, &fmt, patterns) {
None => {
// Row was empty/invalid — skipped (mirrors JS `if (!record) continue`)
}
Some(parsed) => {
match insert_row_2016(conn, &cfg.insert.sql, &parsed, SCORE_FIELDS_2016) {
Ok(()) => file_rows += 1,
Err(e) => {
*total_errors += 1;
if *total_errors <= 5 {
eprintln!(" [warn] {base}: {e}");
}
}
}
}
}
}
}
Ok(file_rows)
}
+197
View File
@@ -0,0 +1,197 @@
/// Spreadsheet reader: wraps calamine to iterate rows across sheets.
///
/// Sheet selection mirrors the JS scripts:
/// - sheet_mode = "all" → iterate every sheet (handles HCM/HN 65k overflow in data/)
/// - sheet_mode = "first" → sheet 0 only (data-old/)
///
/// Header detection mirrors build-lib.js isHeaderRow:
/// row[0].toUpperCase() in {"HO_TEN", "HỌ TÊN", "STT"}
use std::path::Path;
use calamine::{open_workbook_auto, Data, Reader, Sheets};
use crate::config::{DatasetConfig, HeaderCfg, SheetMode};
use crate::error::BuildError;
// ---------------------------------------------------------------------------
// Public row representation from calamine
// ---------------------------------------------------------------------------
pub type RawRow = Vec<Data>;
// ---------------------------------------------------------------------------
// Header detection — mirrors build-lib.js isHeaderRow
// ---------------------------------------------------------------------------
/// Returns true when the first cell (uppercased) matches one of the configured
/// header tokens. Used to skip the header row on the first row of each sheet.
pub fn is_header_row(row: &[Data], header_cfg: &HeaderCfg) -> bool {
if row.len() < 3 {
return false;
}
let first = row[0].to_string().trim().to_uppercase();
header_cfg.tokens.iter().any(|t| t.to_uppercase() == first)
}
// ---------------------------------------------------------------------------
// All-blank row check (data-old2: strip_blank_rows)
// ---------------------------------------------------------------------------
pub fn is_all_blank(row: &[Data]) -> bool {
row.iter()
.all(|c| matches!(c, Data::Empty) || c.to_string().trim().is_empty())
}
// ---------------------------------------------------------------------------
// File processor — yields all data rows from the file
// ---------------------------------------------------------------------------
/// Process one spreadsheet file, calling `on_row` for each data row.
///
/// `on_row` receives `(sheet_index, row_index_in_sheet, raw_row)` where
/// `row_index_in_sheet` is 0-based AFTER the header has been consumed.
/// Returns `(sheets_seen, total_rows_yielded)`.
pub fn process_file<F>(
path: &Path,
cfg: &DatasetConfig,
mut on_row: F,
) -> Result<(usize, usize), BuildError>
where
F: FnMut(usize, &RawRow),
{
let path_str = path.display().to_string();
// calamine::open_workbook_auto dispatches on file extension
let mut workbook: Sheets<_> = open_workbook_auto(path).map_err(|e| BuildError::Calamine {
path: path_str.clone(),
source: e,
})?;
let sheet_names: Vec<String> = workbook.sheet_names().to_vec();
if sheet_names.is_empty() {
return Err(BuildError::NoSheets(path_str.clone()));
}
// Sheet selection per config
let sheets_to_read: Vec<String> = match cfg.reader.sheet_mode {
SheetMode::All => sheet_names.clone(),
SheetMode::First => vec![sheet_names[0].clone()],
};
let mut total_rows = 0usize;
for (sheet_idx, sheet_name) in sheets_to_read.iter().enumerate() {
let range = workbook
.worksheet_range(sheet_name)
.map_err(|e| BuildError::Calamine {
path: path_str.clone(),
source: e,
})?;
let mut first_row = true;
for raw in range.rows() {
let row: RawRow = raw.to_vec();
// Skip header row on first row of each sheet (matches JS: `if (i === 0 && isHeaderRow(...))`)
if first_row {
first_row = false;
if is_header_row(&row, &cfg.header) {
continue;
}
}
on_row(sheet_idx, &row);
total_rows += 1;
}
}
Ok((sheets_to_read.len(), total_rows))
}
// ---------------------------------------------------------------------------
// Unit tests
// ---------------------------------------------------------------------------
#[cfg(test)]
mod tests {
use super::*;
use crate::config::HeaderCfg;
fn hdr(tokens: &[&str]) -> HeaderCfg {
HeaderCfg {
tokens: tokens.iter().map(|s| s.to_string()).collect(),
}
}
#[test]
fn header_detects_ho_ten() {
let row = vec![
Data::String("HO_TEN".into()),
Data::String("NGAY_SINH".into()),
Data::String("SBD".into()),
];
let cfg = hdr(&["HO_TEN", "HỌ TÊN", "STT"]);
assert!(is_header_row(&row, &cfg));
}
#[test]
fn header_detects_stt() {
let row = vec![
Data::String("STT".into()),
Data::String("B".into()),
Data::String("C".into()),
];
let cfg = hdr(&["HO_TEN", "HỌ TÊN", "STT"]);
assert!(is_header_row(&row, &cfg));
}
#[test]
fn header_detects_ho_ten_unicode() {
let row = vec![
Data::String("HỌ TÊN".into()),
Data::String("B".into()),
Data::String("C".into()),
];
let cfg = hdr(&["HO_TEN", "HỌ TÊN", "STT"]);
assert!(is_header_row(&row, &cfg));
}
#[test]
fn header_rejects_data_row() {
let row = vec![
Data::String("Nguyen Van A".into()),
Data::String("01/01/2000".into()),
Data::String("12345678".into()),
];
let cfg = hdr(&["HO_TEN", "HỌ TÊN", "STT"]);
assert!(!is_header_row(&row, &cfg));
}
#[test]
fn header_rejects_short_row() {
let row = vec![Data::String("HO_TEN".into()), Data::Empty];
let cfg = hdr(&["HO_TEN", "HỌ TÊN", "STT"]);
assert!(!is_header_row(&row, &cfg));
}
#[test]
fn header_case_insensitive() {
let row = vec![
Data::String("ho_ten".into()),
Data::String("B".into()),
Data::String("C".into()),
];
let cfg = hdr(&["HO_TEN", "HỌ TÊN", "STT"]);
assert!(is_header_row(&row, &cfg));
}
#[test]
fn blank_row_detection() {
let row = vec![Data::Empty, Data::Empty, Data::String("".into())];
assert!(is_all_blank(&row));
let row2 = vec![Data::String("Nguyen".into()), Data::Empty, Data::Empty];
assert!(!is_all_blank(&row2));
}
}
+410
View File
@@ -0,0 +1,410 @@
/// Row transformation: ascii normalisation, score regex parsing, validation.
///
/// `to_ascii` replicates build-lib.js `toAscii` exactly:
/// str.normalize("NFD").replace(/[̀-ͯ]/g,"").replace(/đ/gi,"d").toLowerCase()
use std::collections::HashMap;
use regex::Regex;
use unicode_normalization::UnicodeNormalization;
use crate::config::{DatasetConfig, ValidationCfg};
use crate::error::BuildError;
// ---------------------------------------------------------------------------
// Compiled score patterns (built once at startup from config)
// ---------------------------------------------------------------------------
pub struct CompiledPatterns {
/// Ordered list so INSERT column order is deterministic
pub patterns: Vec<(String, Regex)>,
}
impl CompiledPatterns {
pub fn new(scores: &HashMap<String, String>) -> Result<Self, BuildError> {
let mut patterns = Vec::with_capacity(scores.len());
for (field, src) in scores {
let re = Regex::new(src).map_err(|e| BuildError::Regex {
pattern: src.clone(),
source: e,
})?;
patterns.push((field.clone(), re));
}
// Sort for deterministic order across HashMap iteration
patterns.sort_by(|a, b| a.0.cmp(&b.0));
Ok(Self { patterns })
}
}
// ---------------------------------------------------------------------------
// to_ascii — must be byte-for-byte equivalent to build-lib.js toAscii
// ---------------------------------------------------------------------------
/// Normalise a Vietnamese name to an ASCII slug.
///
/// Algorithm mirrors the JavaScript `toAscii` in build-lib.js:
/// 1. NFD decompose (splits base + combining diacritics)
/// 2. Drop all Unicode combining marks (U+0300–U+036F)
/// 3. Replace đ/Đ with d (NFD does not decompose đ)
/// 4. Lowercase
pub fn to_ascii(s: &str) -> String {
// Step 1 + 2: NFD then filter out combining marks (Unicode category M)
let decomposed: String = s
.nfd()
.filter(|c| !('\u{0300}'..='\u{036f}').contains(c))
.collect();
// Step 3: đ/Đ are not decomposed by NFD — replace explicitly
let replaced = decomposed.replace(['đ', 'Đ'], "d");
// Step 4: lowercase
replaced.to_lowercase()
}
// ---------------------------------------------------------------------------
// Parsed row ready for DB insert
// ---------------------------------------------------------------------------
pub struct ParsedRow {
pub so_bao_danh: String,
pub ho_ten: String,
pub ho_ten_ascii: String,
pub ngay_sinh: Option<String>,
/// thptqg2016 only: examination cluster name (TEN_CUMTHI column)
pub ten_cum_thi: Option<String>,
/// thptqg2016 only: gender (GIOI_TINH column), normalised to "Nam"/"Nữ" or None
pub gioi_tinh: Option<String>,
/// Subject field → float value; absent subjects not in map → NULL
pub scores: HashMap<String, f64>,
}
// ---------------------------------------------------------------------------
// Row validation — mirrors the per-script skip logic
// ---------------------------------------------------------------------------
/// Returns `None` when the row should be skipped entirely (before sourceRows counter).
/// Returns `Some(reason)` when the row should be counted as sourceRows but skipped.
#[derive(Debug, PartialEq, Eq)]
pub enum SkipReason {
/// Row is fully blank (data-old2 only, before sourceRows counter)
BlankRow,
/// soBaoDanh or hoTen empty/missing
EmptyField,
/// soBaoDanh contains non-digit characters (data-old / data-old2 guard)
NonNumericSbd,
}
/// Validates a raw cell slice against the dataset's `ValidationCfg`.
/// Returns `Ok(())` on pass, `Err(SkipReason)` on fail.
pub fn validate_row(
ho_ten: &str,
so_bao_danh: &str,
cfg: &ValidationCfg,
strip_blank_rows: bool,
all_blank: bool,
) -> Result<(), SkipReason> {
// data-old2: skip fully blank rows BEFORE counting sourceRows
if strip_blank_rows && all_blank {
return Err(SkipReason::BlankRow);
}
if cfg.require_nonempty_sbd && so_bao_danh.is_empty() {
return Err(SkipReason::EmptyField);
}
if cfg.require_nonempty_name && ho_ten.is_empty() {
return Err(SkipReason::EmptyField);
}
if cfg.require_numeric_sbd && !so_bao_danh.chars().all(|c| c.is_ascii_digit()) {
return Err(SkipReason::NonNumericSbd);
}
Ok(())
}
// ---------------------------------------------------------------------------
// Score parsing — mirrors build-lib.js parseScores
// ---------------------------------------------------------------------------
/// Parse a DIEM_THI cell string and extract matching subject scores.
pub fn parse_scores(diem_thi: &str, patterns: &CompiledPatterns) -> HashMap<String, f64> {
let mut out = HashMap::new();
for (field, re) in &patterns.patterns {
if let Some(caps) = re.captures(diem_thi) {
if let Some(m) = caps.get(1) {
if let Ok(v) = m.as_str().parse::<f64>() {
if v.is_finite() {
out.insert(field.clone(), v);
}
}
}
}
}
out
}
// ---------------------------------------------------------------------------
// Full row transform (thptqg2017 fixed-column path)
// ---------------------------------------------------------------------------
/// Extract and transform one spreadsheet row into a `ParsedRow` using fixed column indices.
/// `raw` is the full cell slice; column indices come from `cfg.columns`.
/// Used for thptqg2017 configs that have a static [columns] table.
pub fn transform_row(
raw: &[calamine::Data],
cfg: &DatasetConfig,
patterns: &CompiledPatterns,
) -> ParsedRow {
let cols = cfg
.columns
.as_ref()
.expect("transform_row requires [columns] section in config");
let get = |idx: usize| -> String {
raw.get(idx)
.map(|cell| cell.to_string().trim().to_owned())
.unwrap_or_default()
};
let ho_ten = get(cols.ho_ten);
let ngay_sinh = get(cols.ngay_sinh);
let so_bao_danh = get(cols.so_bao_danh);
let diem_thi = raw
.get(cols.diem_thi)
.map(|c| c.to_string())
.unwrap_or_default();
let ho_ten_ascii = to_ascii(&ho_ten);
let scores = parse_scores(&diem_thi, patterns);
let ngay_sinh_opt = if ngay_sinh.is_empty() {
None
} else {
Some(ngay_sinh)
};
ParsedRow {
so_bao_danh,
ho_ten,
ho_ten_ascii,
ngay_sinh: ngay_sinh_opt,
ten_cum_thi: None,
gioi_tinh: None,
scores,
}
}
// ---------------------------------------------------------------------------
// Unit tests — 20 cases for to_ascii (real Vietnamese names)
// ---------------------------------------------------------------------------
#[cfg(test)]
mod tests {
use super::*;
// Helper: assert to_ascii(input) == expected
fn check(input: &str, expected: &str) {
assert_eq!(
to_ascii(input),
expected,
"to_ascii({input:?}) expected {expected:?}"
);
}
#[test]
fn ascii_plain_latin() {
check("Nguyen Van A", "nguyen van a");
}
#[test]
fn ascii_nguyen_thi_hoa() {
check("Nguyễn Thị Hoa", "nguyen thi hoa");
}
#[test]
fn ascii_tran_van_duc() {
// đ/Đ replacement
check("Trần Văn Đức", "tran van duc");
}
#[test]
fn ascii_le_thi_my_duyen() {
check("Lê Thị Mỹ Duyên", "le thi my duyen");
}
#[test]
fn ascii_pham_thi_lan() {
check("Phạm Thị Lan", "pham thi lan");
}
#[test]
fn ascii_bui_thi_thu() {
check("Bùi Thị Thu", "bui thi thu");
}
#[test]
fn ascii_hoang_van_truong() {
check("Hoàng Văn Trường", "hoang van truong");
}
#[test]
fn ascii_do_thi_ngan() {
// Đ uppercase at start
check("Đỗ Thị Ngân", "do thi ngan");
}
#[test]
fn ascii_nguyen_van_khanh() {
check("Nguyễn Văn Khánh", "nguyen van khanh");
}
#[test]
fn ascii_trinh_thi_bich_ngoc() {
check("Trịnh Thị Bích Ngọc", "trinh thi bich ngoc");
}
#[test]
fn ascii_vu_thi_dieu() {
// ề = e + combining grave + combining circumflex (after NFD)
check("Vũ Thị Diệu", "vu thi dieu");
}
#[test]
fn ascii_nguyen_thi_tuong_vi() {
check("Nguyễn Thị Tường Vi", "nguyen thi tuong vi");
}
#[test]
fn ascii_lowercase_d_stroke() {
// Lowercase đ → d
check("đặng thị hằng", "dang thi hang");
}
#[test]
fn ascii_uppercase_d_stroke() {
check("ĐẶNG THỊ HẰNG", "dang thi hang");
}
#[test]
fn ascii_mixed_case() {
check("NGUYỄN VĂN AN", "nguyen van an");
}
#[test]
fn ascii_tran_thi_kim_anh() {
check("Trần Thị Kim Anh", "tran thi kim anh");
}
#[test]
fn ascii_nguyen_thi_phuong_thao() {
check("Nguyễn Thị Phương Thảo", "nguyen thi phuong thao");
}
#[test]
fn ascii_le_van_long() {
check("Lê Văn Long", "le van long");
}
#[test]
fn ascii_vo_thi_xuan_mai() {
check("Võ Thị Xuân Mai", "vo thi xuan mai");
}
#[test]
fn ascii_empty_string() {
check("", "");
}
// --- Score parsing tests ---
fn make_patterns() -> CompiledPatterns {
let mut map = HashMap::new();
map.insert("toan".into(), r"Toán:\s*(\d+(?:\.\d+)?)".into());
map.insert("ngu_van".into(), r"Ngữ văn:\s*(\d+(?:\.\d+)?)".into());
map.insert("vat_ly".into(), r"Vật lí:\s*(\d+(?:\.\d+)?)".into());
CompiledPatterns::new(&map).unwrap()
}
#[test]
fn parse_scores_single() {
let p = make_patterns();
let s = "Toán: 8.5";
let scores = parse_scores(s, &p);
assert_eq!(scores.get("toan"), Some(&8.5));
assert!(scores.get("ngu_van").is_none());
}
#[test]
fn parse_scores_multiple() {
let p = make_patterns();
let s = "Toán: 7.25 Ngữ văn: 6.0 Vật lí: 9";
let scores = parse_scores(s, &p);
assert_eq!(scores.get("toan"), Some(&7.25));
assert_eq!(scores.get("ngu_van"), Some(&6.0));
assert_eq!(scores.get("vat_ly"), Some(&9.0));
}
#[test]
fn parse_scores_empty_cell() {
let p = make_patterns();
let scores = parse_scores("", &p);
assert!(scores.is_empty());
}
// --- Validation tests ---
fn default_validation() -> ValidationCfg {
ValidationCfg {
require_numeric_sbd: false,
require_nonempty_name: true,
require_nonempty_sbd: true,
}
}
#[test]
fn validate_ok() {
let v = default_validation();
assert!(validate_row("Nguyen Van A", "12345678", &v, false, false).is_ok());
}
#[test]
fn validate_empty_sbd() {
let v = default_validation();
assert_eq!(
validate_row("Nguyen Van A", "", &v, false, false),
Err(SkipReason::EmptyField)
);
}
#[test]
fn validate_empty_name() {
let v = default_validation();
assert_eq!(
validate_row("", "12345678", &v, false, false),
Err(SkipReason::EmptyField)
);
}
#[test]
fn validate_non_numeric_sbd_rejected() {
let mut v = default_validation();
v.require_numeric_sbd = true;
assert_eq!(
validate_row("Nguyen Van A", "12AB5678", &v, false, false),
Err(SkipReason::NonNumericSbd)
);
}
#[test]
fn validate_numeric_sbd_accepted() {
let mut v = default_validation();
v.require_numeric_sbd = true;
assert!(validate_row("Nguyen Van A", "12345678", &v, false, false).is_ok());
}
#[test]
fn validate_blank_row_skipped() {
let v = default_validation();
// strip_blank_rows=true AND all_blank=true → BlankRow
assert_eq!(
validate_row("", "", &v, true, true),
Err(SkipReason::BlankRow)
);
}
}
+201
View File
@@ -0,0 +1,201 @@
/// SQLite writer: DDL setup, batched INSERT OR REPLACE, VACUUM, stats output.
///
/// Mirrors build-lib.js createDb + the transaction loop in each build-database*.js.
/// Stats output lines match the JS stdout exactly so existing CI log-greps still work.
///
/// thptqg2016 differences vs thptqg2017:
/// - INSERT includes ten_cum_thi and gioi_tinh after ngay_sinh
/// - Score columns are toan/ngu_van/vat_ly/hoa_hoc/sinh_hoc/lich_su/dia_ly/
/// tieng_anh/tieng_phap/tieng_duc/tieng_nhat/tieng_trung
/// (no khtn, khxh, tieng_nga)
use std::fs;
use std::path::Path;
use rusqlite::{params_from_iter, Connection, ToSql};
use crate::config::DatasetConfig;
use crate::error::BuildError;
use crate::transform::ParsedRow;
// ---------------------------------------------------------------------------
// DB initialisation — mirrors build-lib.js createDb (delete + recreate)
// ---------------------------------------------------------------------------
/// Open (or recreate) the output SQLite database, execute the DDL from config,
/// and return the open connection ready for inserts.
pub fn open_db(db_path: &Path, cfg: &DatasetConfig) -> Result<Connection, BuildError> {
// Mirror Node behaviour: delete existing file before creating (build-lib.js:54)
if db_path.exists() {
fs::remove_file(db_path).map_err(|e| BuildError::Io {
path: db_path.display().to_string(),
source: e,
})?;
}
// Ensure parent directory exists
if let Some(parent) = db_path.parent() {
if !parent.as_os_str().is_empty() {
fs::create_dir_all(parent).map_err(|e| BuildError::Io {
path: parent.display().to_string(),
source: e,
})?;
}
}
let conn = Connection::open(db_path)?;
conn.execute_batch(&cfg.schema.ddl)?;
Ok(conn)
}
// ---------------------------------------------------------------------------
// Score field lists — one per dataset schema
// ---------------------------------------------------------------------------
/// thptqg2017 score columns (14 fields including khtn/khxh/tieng_nga).
pub const SCORE_FIELDS_2017: &[&str] = &[
"toan",
"ngu_van",
"vat_ly",
"hoa_hoc",
"sinh_hoc",
"khtn",
"lich_su",
"dia_ly",
"gdcd",
"khxh",
"tieng_anh",
"tieng_phap",
"tieng_nga",
"tieng_trung",
];
/// thptqg2016 score columns (12 fields: tieng_duc/tieng_nhat present; no khtn/khxh/tieng_nga).
pub const SCORE_FIELDS_2016: &[&str] = &[
"toan",
"ngu_van",
"vat_ly",
"hoa_hoc",
"sinh_hoc",
"lich_su",
"dia_ly",
"tieng_anh",
"tieng_phap",
"tieng_duc",
"tieng_nhat",
"tieng_trung",
];
/// Alias kept so thptqg2017 callers that import SCORE_FIELDS continue to compile.
pub const SCORE_FIELDS: &[&str] = SCORE_FIELDS_2017;
// ---------------------------------------------------------------------------
// Insert a single parsed row (thptqg2017 schema — no ten_cum_thi / gioi_tinh)
// ---------------------------------------------------------------------------
/// Bind all fields from `row` into the prepared statement and execute it.
/// Positional params: so_bao_danh, ho_ten, ho_ten_ascii, ngay_sinh, <scores...>
pub fn insert_row(
conn: &Connection,
sql: &str,
row: &ParsedRow,
score_fields: &[&str],
) -> Result<(), BuildError> {
let mut params: Vec<Box<dyn ToSql>> = Vec::with_capacity(4 + score_fields.len());
params.push(Box::new(row.so_bao_danh.clone()));
params.push(Box::new(row.ho_ten.clone()));
params.push(Box::new(row.ho_ten_ascii.clone()));
params.push(Box::new(row.ngay_sinh.clone()));
for field in score_fields {
let val: Option<f64> = row.scores.get(*field).copied();
params.push(Box::new(val));
}
conn.execute(sql, params_from_iter(params.iter().map(|p| p.as_ref())))?;
Ok(())
}
// ---------------------------------------------------------------------------
// Insert a single parsed row (thptqg2016 schema — includes ten_cum_thi + gioi_tinh)
// ---------------------------------------------------------------------------
/// Bind all fields from `row` into the prepared statement for the thptqg2016 schema.
/// Positional params: so_bao_danh, ho_ten, ho_ten_ascii, ngay_sinh, ten_cum_thi,
/// gioi_tinh, <scores...>
pub fn insert_row_2016(
conn: &Connection,
sql: &str,
row: &ParsedRow,
score_fields: &[&str],
) -> Result<(), BuildError> {
let mut params: Vec<Box<dyn ToSql>> = Vec::with_capacity(6 + score_fields.len());
params.push(Box::new(row.so_bao_danh.clone()));
params.push(Box::new(row.ho_ten.clone()));
params.push(Box::new(row.ho_ten_ascii.clone()));
params.push(Box::new(row.ngay_sinh.clone()));
params.push(Box::new(row.ten_cum_thi.clone()));
params.push(Box::new(row.gioi_tinh.clone()));
for field in score_fields {
let val: Option<f64> = row.scores.get(*field).copied();
params.push(Box::new(val));
}
conn.execute(sql, params_from_iter(params.iter().map(|p| p.as_ref())))?;
Ok(())
}
// ---------------------------------------------------------------------------
// Post-build: VACUUM + stats output
// ---------------------------------------------------------------------------
/// Run VACUUM and print statistics lines that mirror the Node scripts' stdout.
/// The exact prefix tokens ("Source data rows", "DB rows", "Size:") are preserved
/// so any log-grep in the deploy pipeline keeps working.
#[allow(clippy::too_many_arguments)]
pub fn finish_db(
conn: &Connection,
db_path: &Path,
source_rows: u64,
skipped: u64,
errors: u64,
dataset_label: &str,
_file_count: usize,
is_old2: bool,
) -> Result<(), BuildError> {
conn.execute_batch("VACUUM")?;
let db_count: i64 = conn.query_row("SELECT COUNT(*) FROM student", [], |row| row.get(0))?;
let insertable = source_rows - skipped;
println!();
if is_old2 {
println!("Source non-blank data rows: {source_rows}");
println!(" skipped (empty/non-numeric SBD): {skipped}");
} else {
println!("Source data rows (post-header): {source_rows}");
if dataset_label.contains("old") {
println!(" skipped (empty/non-numeric SBD): {skipped}");
} else {
println!(" skipped (empty/invalid): {skipped}");
}
}
println!(" insertable: {insertable}");
println!(" insert errors: {errors}");
println!("DB rows (distinct SBD): {db_count}");
if !dataset_label.contains("old") && errors == 0 {
let gap = insertable as i64 - db_count;
if gap == 0 {
println!("Audit: OK — every source row made it in.");
} else {
println!("Audit: {gap} row(s) collapsed (duplicate SBDs overwriting).");
}
}
let sz = fs::metadata(db_path).map(|m| m.len()).unwrap_or(0);
println!("Size: {:.1} MB", sz as f64 / 1024.0 / 1024.0);
Ok(())
}
+18
View File
@@ -0,0 +1,18 @@
# Test Fixtures
Anonymised `.xlsx` files for integration testing. All student PII has been replaced:
- `ho_ten` replaced with `Nguyen Van Test NNN` / `Tran Thi Test NNN` patterns
- `so_bao_danh` replaced with sequential synthetic numbers (e.g. `10000001`)
- `ngay_sinh` replaced with fixed synthetic dates
- Scores are realistic random values in the 0–10 range
Files:
- `province-100.xlsx` — 100-row single-sheet file (simulates a normal province)
- `hcm-overflow.xlsx` — 2-sheet file (200 rows Sheet1 + 200 rows Sheet2, simulating HCM overflow)
- `province-numeric-sbd.xlsx` — 100 rows with strictly numeric SBDs (for data-old variant)
These files are generated by `tests/golden.rs` `generate_fixtures()` if they do not already exist
on disk. The generator is pure Rust (uses the `zip` crate already pulled in via calamine).
No external Python or Node tooling required for unit/integration tests.
Binary file not shown.
Binary file not shown.
Binary file not shown.
+691
View File
@@ -0,0 +1,691 @@
/// Stage 5 golden tests — integration tests using anonymised fixture files.
///
/// Fixture files are generated in-process via raw OOXML + zip if they do not
/// already exist on disk. No external Python or Node tooling required for the
/// Rust-side tests. The Node golden comparison is marked #[ignore] when pnpm
/// is not in PATH.
use std::io::Write as IoWrite;
use std::path::{Path, PathBuf};
// ---------------------------------------------------------------------------
// Minimal OOXML xlsx generator
//
// Produces a valid .xlsx that calamine can read. Only uses the `zip` crate
// which is already pulled in as a transitive dependency of calamine.
// ---------------------------------------------------------------------------
/// One row of cell data for a fixture sheet.
struct XlsxRow {
values: Vec<String>,
}
/// Write a minimal .xlsx to `path` with the given sheets.
/// `sheets`: Vec<(sheet_name, rows)> where rows[0] is the header.
fn write_xlsx(path: &Path, sheets: &[(String, Vec<XlsxRow>)]) {
use zip::{write::SimpleFileOptions, ZipWriter};
let file = std::fs::File::create(path).expect("create fixture xlsx");
let mut zip = ZipWriter::new(file);
let opts = SimpleFileOptions::default();
// [Content_Types].xml
let mut content_types = String::from(
r#"<?xml version="1.0" encoding="UTF-8" standalone="yes"?>
<Types xmlns="http://schemas.openxmlformats.org/package/2006/content-types">
<Default Extension="rels" ContentType="application/vnd.openxmlformats-package.relationships+xml"/>
<Default Extension="xml" ContentType="application/xml"/>
<Override PartName="/xl/workbook.xml" ContentType="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet.main+xml"/>
"#,
);
for (i, _) in sheets.iter().enumerate() {
content_types.push_str(&format!(
r#" <Override PartName="/xl/worksheets/sheet{}.xml" ContentType="application/vnd.openxmlformats-officedocument.spreadsheetml.worksheet+xml"/>
"#,
i + 1
));
}
content_types.push_str("</Types>");
zip.start_file("[Content_Types].xml", opts).unwrap();
zip.write_all(content_types.as_bytes()).unwrap();
// _rels/.rels
zip.start_file("_rels/.rels", opts).unwrap();
zip.write_all(
br#"<?xml version="1.0" encoding="UTF-8" standalone="yes"?>
<Relationships xmlns="http://schemas.openxmlformats.org/package/2006/relationships">
<Relationship Id="rId1" Type="http://schemas.openxmlformats.org/officeDocument/2006/relationships/officeDocument" Target="xl/workbook.xml"/>
</Relationships>"#,
)
.unwrap();
// xl/_rels/workbook.xml.rels
let mut wb_rels = String::from(
r#"<?xml version="1.0" encoding="UTF-8" standalone="yes"?>
<Relationships xmlns="http://schemas.openxmlformats.org/package/2006/relationships">
"#,
);
for (i, _) in sheets.iter().enumerate() {
wb_rels.push_str(&format!(
r#" <Relationship Id="rId{}" Type="http://schemas.openxmlformats.org/officeDocument/2006/relationships/worksheet" Target="worksheets/sheet{}.xml"/>
"#,
i + 1,
i + 1
));
}
wb_rels.push_str("</Relationships>");
zip.start_file("xl/_rels/workbook.xml.rels", opts).unwrap();
zip.write_all(wb_rels.as_bytes()).unwrap();
// xl/workbook.xml
let mut wb = String::from(
r#"<?xml version="1.0" encoding="UTF-8" standalone="yes"?>
<workbook xmlns="http://schemas.openxmlformats.org/spreadsheetml/2006/main"
xmlns:r="http://schemas.openxmlformats.org/officeDocument/2006/relationships">
<sheets>
"#,
);
for (i, (name, _)) in sheets.iter().enumerate() {
let escaped = xml_escape(name);
wb.push_str(&format!(
r#" <sheet name="{}" sheetId="{}" r:id="rId{}"/>
"#,
escaped,
i + 1,
i + 1
));
}
wb.push_str(" </sheets>\n</workbook>");
zip.start_file("xl/workbook.xml", opts).unwrap();
zip.write_all(wb.as_bytes()).unwrap();
// xl/worksheets/sheetN.xml
for (i, (_, rows)) in sheets.iter().enumerate() {
let mut ws = String::from(
r#"<?xml version="1.0" encoding="UTF-8" standalone="yes"?>
<worksheet xmlns="http://schemas.openxmlformats.org/spreadsheetml/2006/main">
<sheetData>
"#,
);
for (row_idx, row) in rows.iter().enumerate() {
ws.push_str(&format!(
r#" <row r="{}">
"#,
row_idx + 1
));
for (col_idx, val) in row.values.iter().enumerate() {
let col_letter = col_letter(col_idx);
let cell_ref = format!("{}{}", col_letter, row_idx + 1);
let escaped = xml_escape(val);
ws.push_str(&format!(
r#" <c r="{}" t="inlineStr"><is><t>{}</t></is></c>
"#,
cell_ref, escaped
));
}
ws.push_str(" </row>\n");
}
ws.push_str(" </sheetData>\n</worksheet>");
zip.start_file(&format!("xl/worksheets/sheet{}.xml", i + 1), opts)
.unwrap();
zip.write_all(ws.as_bytes()).unwrap();
}
zip.finish().unwrap();
}
fn col_letter(idx: usize) -> &'static str {
const LETTERS: &[&str] = &[
"A", "B", "C", "D", "E", "F", "G", "H", "I", "J", "K", "L", "M", "N", "O", "P", "Q", "R",
"S", "T", "U", "V", "W", "X", "Y", "Z",
];
LETTERS[idx % 26]
}
fn xml_escape(s: &str) -> String {
s.replace('&', "&amp;")
.replace('<', "&lt;")
.replace('>', "&gt;")
.replace('"', "&quot;")
}
// ---------------------------------------------------------------------------
// Fixture data builders
// ---------------------------------------------------------------------------
fn header_row() -> XlsxRow {
XlsxRow {
values: vec![
"HO_TEN".into(),
"NGAY_SINH".into(),
"SO_BAO_DANH".into(),
"DIEM_THI".into(),
],
}
}
fn data_row(idx: usize, scores: &str) -> XlsxRow {
// Anonymised: name uses sequential pattern, SBD is purely synthetic
let name = if idx % 2 == 0 {
format!("Nguyen Van Test {:03}", idx)
} else {
format!("Tran Thi Test {:03}", idx)
};
XlsxRow {
values: vec![
name,
format!("15/0{}/{}", (idx % 9) + 1, 1999 + (idx % 5)),
format!("1000{:04}", idx),
scores.to_owned(),
],
}
}
fn sample_scores(idx: usize) -> String {
// Realistic scores in 0–10 range, varies by idx
let toan = 4.0 + (idx % 60) as f64 / 10.0;
let van = 3.5 + (idx % 65) as f64 / 10.0;
format!("Toán: {toan:.1} Ngữ văn: {van:.1} Tiếng Anh: 7.5")
}
// ---------------------------------------------------------------------------
// Fixture file paths
// ---------------------------------------------------------------------------
fn fixtures_dir() -> PathBuf {
// tests/fixtures/ relative to the crate root
let mut p = PathBuf::from(env!("CARGO_MANIFEST_DIR"));
p.push("tests");
p.push("fixtures");
p
}
fn province_fixture_path() -> PathBuf {
fixtures_dir().join("province-100.xlsx")
}
fn hcm_overflow_fixture_path() -> PathBuf {
fixtures_dir().join("hcm-overflow.xlsx")
}
fn numeric_sbd_fixture_path() -> PathBuf {
fixtures_dir().join("province-numeric-sbd.xlsx")
}
// ---------------------------------------------------------------------------
// Fixture generation — called once per test run if files missing
// ---------------------------------------------------------------------------
fn ensure_fixtures() {
let dir = fixtures_dir();
std::fs::create_dir_all(&dir).expect("create fixtures dir");
// province-100.xlsx — 100 data rows, single sheet, with header
if !province_fixture_path().exists() {
let mut rows = vec![header_row()];
for i in 0..100 {
rows.push(data_row(i, &sample_scores(i)));
}
write_xlsx(&province_fixture_path(), &[("Sheet1".to_owned(), rows)]);
}
// hcm-overflow.xlsx — 2 sheets × 200 rows each (no header on sheet 2)
if !hcm_overflow_fixture_path().exists() {
let mut sheet1 = vec![header_row()];
for i in 0..200 {
sheet1.push(data_row(i, &sample_scores(i)));
}
// Sheet2: continuation rows, no header row (as in real HCM overflow)
let mut sheet2 = Vec::new();
for i in 200..400 {
sheet2.push(data_row(i, &sample_scores(i)));
}
write_xlsx(
&hcm_overflow_fixture_path(),
&[("Sheet1".to_owned(), sheet1), ("Sheet2".to_owned(), sheet2)],
);
}
// province-numeric-sbd.xlsx — strictly numeric SBDs for data-old config
if !numeric_sbd_fixture_path().exists() {
let mut rows = vec![header_row()];
for i in 0..100 {
rows.push(XlsxRow {
values: vec![
format!("Nguyen Van Test {:03}", i),
"01/01/2000".to_owned(),
format!("{:08}", 20000000 + i), // pure digits
sample_scores(i),
],
});
}
write_xlsx(&numeric_sbd_fixture_path(), &[("Sheet1".to_owned(), rows)]);
}
}
// ---------------------------------------------------------------------------
// Config helpers
// ---------------------------------------------------------------------------
fn make_data_config() -> xlsxread::config::DatasetConfig {
let cfg_path = PathBuf::from(env!("CARGO_MANIFEST_DIR"))
.join("configs")
.join("thptqg2017-data.toml");
xlsxread::config::load_config(&cfg_path).expect("load data config")
}
// ---------------------------------------------------------------------------
// Integration tests — pure Rust, no Node dependency
// ---------------------------------------------------------------------------
#[test]
fn province_100_builds_100_rows() {
ensure_fixtures();
let dir = tempdir();
let db_path = dir.join("test.db");
let fixture_dir = dir.join("input");
std::fs::create_dir_all(&fixture_dir).unwrap();
std::fs::copy(province_fixture_path(), fixture_dir.join("province.xlsx")).unwrap();
run_build_cmd(&fixture_dir, &db_path, "thptqg2017-data.toml");
let count = query_count(&db_path);
assert_eq!(count, 100, "expected 100 rows from province-100 fixture");
}
#[test]
fn hcm_overflow_builds_400_rows() {
ensure_fixtures();
let dir = tempdir();
let db_path = dir.join("test.db");
let fixture_dir = dir.join("input");
std::fs::create_dir_all(&fixture_dir).unwrap();
std::fs::copy(hcm_overflow_fixture_path(), fixture_dir.join("hcm.xlsx")).unwrap();
run_build_cmd(&fixture_dir, &db_path, "thptqg2017-data.toml");
let count = query_count(&db_path);
assert_eq!(
count, 400,
"expected 400 rows (200 × 2 sheets) from hcm-overflow fixture"
);
}
#[test]
fn data_old_first_sheet_only_100_rows() {
ensure_fixtures();
let dir = tempdir();
let db_path = dir.join("test.db");
let fixture_dir = dir.join("input");
std::fs::create_dir_all(&fixture_dir).unwrap();
// Use the overflow file but with data-old config (first sheet only → 200 rows)
std::fs::copy(hcm_overflow_fixture_path(), fixture_dir.join("hcm.xlsx")).unwrap();
run_build_cmd(&fixture_dir, &db_path, "thptqg2017-data-old.toml");
// data-old: sheet_mode=first → only 200 rows from sheet1; but SBDs "1000NNNN" are
// all digits so all pass the numeric guard
let count = query_count(&db_path);
assert_eq!(
count, 200,
"data-old config should read only first sheet (200 rows)"
);
}
#[test]
fn numeric_sbd_guard_rejects_non_numeric() {
ensure_fixtures();
let dir = tempdir();
let db_path = dir.join("test.db");
let fixture_dir = dir.join("input");
std::fs::create_dir_all(&fixture_dir).unwrap();
// Write a fixture with one non-numeric SBD mixed in
let mut rows = vec![header_row()];
for i in 0..10 {
rows.push(XlsxRow {
values: vec![
format!("Test {:03}", i),
"01/01/2000".to_owned(),
if i == 5 {
"ABC123".to_owned()
} else {
format!("{:08}", 20000000 + i)
},
sample_scores(i),
],
});
}
let mixed_path = fixture_dir.join("mixed.xlsx");
write_xlsx(&mixed_path, &[("Sheet1".to_owned(), rows)]);
run_build_cmd(&fixture_dir, &db_path, "thptqg2017-data-old.toml");
// Row i=5 has non-numeric SBD → rejected by data-old config
let count = query_count(&db_path);
assert_eq!(
count, 9,
"non-numeric SBD row should be skipped by data-old config"
);
}
#[test]
fn scores_parsed_correctly_into_db() {
ensure_fixtures();
let dir = tempdir();
let db_path = dir.join("test.db");
let fixture_dir = dir.join("input");
std::fs::create_dir_all(&fixture_dir).unwrap();
let rows = vec![
header_row(),
XlsxRow {
values: vec![
"Nguyen Van Test 001".to_owned(),
"01/01/2000".to_owned(),
"10000001".to_owned(),
"Toán: 8.5 Ngữ văn: 7.0 Tiếng Anh: 9.25".to_owned(),
],
},
];
write_xlsx(
&fixture_dir.join("one.xlsx"),
&[("Sheet1".to_owned(), rows)],
);
run_build_cmd(&fixture_dir, &db_path, "thptqg2017-data.toml");
let conn = rusqlite::Connection::open(&db_path).unwrap();
let (toan, van, anh): (f64, f64, f64) = conn
.query_row(
"SELECT toan, ngu_van, tieng_anh FROM student WHERE so_bao_danh = '10000001'",
[],
|r| Ok((r.get(0)?, r.get(1)?, r.get(2)?)),
)
.expect("row not found");
assert!((toan - 8.5).abs() < 1e-9);
assert!((van - 7.0).abs() < 1e-9);
assert!((anh - 9.25).abs() < 1e-9);
}
#[test]
fn to_ascii_stored_correctly() {
ensure_fixtures();
let dir = tempdir();
let db_path = dir.join("test.db");
let fixture_dir = dir.join("input");
std::fs::create_dir_all(&fixture_dir).unwrap();
let rows = vec![
header_row(),
XlsxRow {
values: vec![
"Nguyễn Văn Đức".to_owned(),
"".to_owned(),
"20000001".to_owned(),
"".to_owned(),
],
},
];
write_xlsx(
&fixture_dir.join("one.xlsx"),
&[("Sheet1".to_owned(), rows)],
);
run_build_cmd(&fixture_dir, &db_path, "thptqg2017-data.toml");
let conn = rusqlite::Connection::open(&db_path).unwrap();
let ascii: String = conn
.query_row(
"SELECT ho_ten_ascii FROM student WHERE so_bao_danh = '20000001'",
[],
|r| r.get(0),
)
.expect("row not found");
assert_eq!(ascii, "nguyen van duc");
}
#[test]
fn audit_subcommand_matches_after_build() {
ensure_fixtures();
let dir = tempdir();
let db_path = dir.join("test.db");
let fixture_dir = dir.join("input");
std::fs::create_dir_all(&fixture_dir).unwrap();
std::fs::copy(province_fixture_path(), fixture_dir.join("province.xlsx")).unwrap();
run_build_cmd(&fixture_dir, &db_path, "thptqg2017-data.toml");
// audit should match (100 distinct SBDs in xlsx == 100 rows in DB)
let cfg = make_data_config();
let result = xlsxread::audit::run_audit(&fixture_dir, &db_path, &cfg).expect("audit failed");
assert!(result.matched, "audit should match after build");
assert_eq!(result.distinct_sbds, 100);
assert_eq!(result.db_count, 100);
}
#[test]
fn audit_subcommand_mismatch_detected() {
ensure_fixtures();
let dir = tempdir();
let db_path = dir.join("test.db");
let fixture_dir = dir.join("input");
std::fs::create_dir_all(&fixture_dir).unwrap();
// Write 10 rows to xlsx but build DB from only 5 rows
let mut all_rows = vec![header_row()];
for i in 0..10 {
all_rows.push(data_row(i, &sample_scores(i)));
}
write_xlsx(
&fixture_dir.join("all.xlsx"),
&[("Sheet1".to_owned(), all_rows)],
);
// Build DB with only first 5 rows in a different file
let build_dir = dir.join("build_input");
std::fs::create_dir_all(&build_dir).unwrap();
let mut five_rows = vec![header_row()];
for i in 0..5 {
five_rows.push(data_row(i, &sample_scores(i)));
}
write_xlsx(
&build_dir.join("five.xlsx"),
&[("Sheet1".to_owned(), five_rows)],
);
run_build_cmd(&build_dir, &db_path, "thptqg2017-data.toml");
// audit against fixture_dir (10 xlsx rows) but DB has 5 rows → mismatch
let cfg = make_data_config();
let result = xlsxread::audit::run_audit(&fixture_dir, &db_path, &cfg).expect("audit failed");
assert!(!result.matched, "audit should not match (10 xlsx vs 5 db)");
assert_eq!(result.distinct_sbds, 10);
assert_eq!(result.db_count, 5);
}
// ---------------------------------------------------------------------------
// Golden test: compare Rust DB vs Node DB on identical fixture
// Marked #[ignore] when pnpm / node is not in PATH — CI installs them first.
// ---------------------------------------------------------------------------
#[test]
#[ignore]
fn golden_rust_matches_node_db() {
// This test requires: pnpm, node, and the thptqg2017 package to be installed
// Run with: cargo test -- --ignored golden_rust_matches_node_db
let which_pnpm = std::process::Command::new("which")
.arg("pnpm")
.output()
.map(|o| o.status.success())
.unwrap_or(false);
if !which_pnpm {
eprintln!("pnpm not in PATH — skipping golden test");
return;
}
let dir = tempdir();
let fixture_dir = dir.join("input");
std::fs::create_dir_all(&fixture_dir).unwrap();
std::fs::copy(province_fixture_path(), fixture_dir.join("province.xlsx")).unwrap();
// Build with Rust
let rust_db = dir.join("rust.db");
run_build_cmd(&fixture_dir, &rust_db, "thptqg2017-data.toml");
// Build with Node (run build-database.js with DATA_DIR / DB_PATH overrides)
// Node script reads env-vars via a thin wrapper — see scripts/build-database.js
// For now: diff via SELECT * ORDER BY so_bao_danh
let node_db = dir.join("node.db");
let status = std::process::Command::new("pnpm")
.args(["exec", "node", "scripts/build-database.js"])
.env("OVERRIDE_SRC_DIR", fixture_dir.to_str().unwrap())
.env("OVERRIDE_DB_PATH", node_db.to_str().unwrap())
.current_dir(
PathBuf::from(env!("CARGO_MANIFEST_DIR"))
.parent()
.unwrap()
.parent()
.unwrap(),
)
.status()
.expect("failed to run node build script");
if !status.success() {
panic!("Node build script failed with: {status}");
}
// Row-by-row comparison
diff_dbs(&rust_db, &node_db);
}
// ---------------------------------------------------------------------------
// Helpers
// ---------------------------------------------------------------------------
fn tempdir() -> PathBuf {
let base = std::env::temp_dir().join(format!(
"xlsxread-test-{}",
std::time::SystemTime::now()
.duration_since(std::time::UNIX_EPOCH)
.unwrap()
.subsec_nanos()
));
std::fs::create_dir_all(&base).unwrap();
base
}
fn run_build_cmd(input_dir: &Path, db_path: &Path, config_name: &str) {
let cfg_path = PathBuf::from(env!("CARGO_MANIFEST_DIR"))
.join("configs")
.join(config_name);
let cfg = xlsxread::config::load_config(&cfg_path)
.unwrap_or_else(|e| panic!("load config {config_name}: {e}"));
let patterns =
xlsxread::transform::CompiledPatterns::new(&cfg.scores).expect("compile patterns");
// Collect files
let mut files: Vec<PathBuf> = std::fs::read_dir(input_dir)
.unwrap()
.filter_map(|e| e.ok())
.map(|e| e.path())
.filter(|p| {
p.is_file()
&& p.extension()
.and_then(|e| e.to_str())
.map(|e| {
let l = e.to_lowercase();
l == "xls" || l == "xlsx"
})
.unwrap_or(false)
})
.collect();
files.sort();
let conn = xlsxread::writer::open_db(db_path, &cfg).expect("open db");
conn.execute_batch("BEGIN").unwrap();
for file in &files {
xlsxread::reader::process_file(file, &cfg, |_, raw| {
let all_blank = xlsxread::reader::is_all_blank(raw);
if cfg.reader.strip_blank_rows && all_blank {
return;
}
let cols = cfg.columns.as_ref().expect("golden test requires [columns]");
let ho_ten = raw
.get(cols.ho_ten)
.map(|c| c.to_string().trim().to_owned())
.unwrap_or_default();
let so_bao_danh = raw
.get(cols.so_bao_danh)
.map(|c| c.to_string().trim().to_owned())
.unwrap_or_default();
if xlsxread::transform::validate_row(
&ho_ten,
&so_bao_danh,
&cfg.validation,
cfg.reader.strip_blank_rows,
all_blank,
)
.is_err()
{
return;
}
let row = xlsxread::transform::transform_row(raw, &cfg, &patterns);
let _ = xlsxread::writer::insert_row(
&conn,
&cfg.insert.sql,
&row,
xlsxread::writer::SCORE_FIELDS,
);
})
.expect("process file");
}
conn.execute_batch("COMMIT").unwrap();
conn.execute_batch("VACUUM").unwrap();
}
fn query_count(db_path: &Path) -> i64 {
let conn = rusqlite::Connection::open(db_path).expect("open db for count");
conn.query_row("SELECT COUNT(*) FROM student", [], |r| r.get(0))
.expect("count query")
}
fn diff_dbs(a: &Path, b: &Path) {
let conn_a = rusqlite::Connection::open(a).unwrap();
// Attach b as "other"
conn_a
.execute_batch(&format!("ATTACH DATABASE '{}' AS other", b.display()))
.unwrap();
// Rows in a not in b
let missing_in_b: i64 = conn_a
.query_row(
"SELECT COUNT(*) FROM main.student s
WHERE NOT EXISTS (SELECT 1 FROM other.student o WHERE o.so_bao_danh = s.so_bao_danh)",
[],
|r| r.get(0),
)
.unwrap();
// Rows in b not in a
let missing_in_a: i64 = conn_a
.query_row(
"SELECT COUNT(*) FROM other.student o
WHERE NOT EXISTS (SELECT 1 FROM main.student s WHERE s.so_bao_danh = o.so_bao_danh)",
[],
|r| r.get(0),
)
.unwrap();
assert_eq!(
missing_in_b, 0,
"{missing_in_b} rows in Rust DB missing from Node DB"
);
assert_eq!(
missing_in_a, 0,
"{missing_in_a} rows in Node DB missing from Rust DB"
);
}