mirror of
https://github.com/tiennm99/thptqg.git
synced 2026-10-11 03:13:48 +00:00
feat: replace xlsx (SheetJS) build pipeline with Rust xlsxread CLI (#1)
* feat(xlsxread): vendor Rust binary cloned from thptqg2017 Copies the xlsxread Rust crate from thptqg2017@8b4a755 (chore/xlsxread-rust). Adds format_detect_2016 module with per-file column-layout auto-detection mirroring detectFormat() in scripts/build-database.js (lines 63-87): - separate-scores: SBD/HOTEN/TOAN... fixed columns (dhhanghai files) - mapped: header-derived SOBAODANH|SBD + DIEM_THI dynamic indices - default: positional 6-col layout (no header) Extends ParsedRow with ten_cum_thi and gioi_tinh fields. Adds SCORE_FIELDS_2016 (12 cols: tieng_duc/tieng_nhat; no khtn/khxh/tieng_nga). Adds thptqg2016-data.toml config with 18-column schema and format_detection flag. 58 tests pass (50 unit + 8 integration), 0 failures. * feat(build): wire build:db to xlsxread CLI Replaces the Node.js build:db script with the xlsxread Rust binary. Adds build:rust script for the cargo compile step in isolation. * chore: remove deprecated build-database.js Superseded by the xlsxread Rust binary. All 119 source files (4 .xls + 115 .xlsx) are now processed by xlsxread with per-file format detection. * ci: build xlsxread before running database build job Adds dtolnay/rust-toolchain@stable and Swatinem/rust-cache@v2 steps before the xlsxread build and database generation steps. Node/pnpm steps now follow the Rust build rather than preceding it. * chore: add Rust build artifacts to .gitignore * docs: update README build instructions for xlsxread pipeline * chore(deps): drop xlsx and better-sqlite3 from package.json and lockfile
This commit is contained in:
1 parent
99465fda59
commit
6b21c625ac
27 files changed
+4422
-315
No files matched your search
Vendored
+16
-3
@@ -20,6 +20,22 @@ jobs:
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- uses: dtolnay/rust-toolchain@stable
|
||||
|
||||
- uses: Swatinem/rust-cache@v2
|
||||
with:
|
||||
workspaces: tools/xlsxread
|
||||
|
||||
- name: Build xlsxread binary
|
||||
run: cargo build --release --manifest-path tools/xlsxread/Cargo.toml
|
||||
|
||||
- name: Build database
|
||||
run: >
|
||||
./tools/xlsxread/target/release/xlsxread build
|
||||
--schema tools/xlsxread/configs/thptqg2016-data.toml
|
||||
--input data
|
||||
--output public/thptqg2016.db
|
||||
|
||||
- uses: pnpm/action-setup@v4
|
||||
|
||||
- uses: actions/setup-node@v4
|
||||
@@ -30,9 +46,6 @@ jobs:
|
||||
- name: Install dependencies
|
||||
run: pnpm install --frozen-lockfile
|
||||
|
||||
- name: Build database
|
||||
run: pnpm build:db
|
||||
|
||||
- name: Compress database
|
||||
run: gzip -k -9 public/thptqg2016.db
|
||||
|
||||
|
||||
@@ -141,3 +141,6 @@ vite.config.ts.timestamp-*
|
||||
# Generated database files
|
||||
public/*.db
|
||||
public/*.db.gz
|
||||
|
||||
# Rust build artifacts
|
||||
tools/xlsxread/target/
|
||||
+22
-7
@@ -19,18 +19,33 @@ Fully static app running entirely in the browser (SQLite via `sql.js`). No backe
|
||||
## Development
|
||||
|
||||
```bash
|
||||
npm install
|
||||
npm run build:db # Parse data/*.xlsx → public/thptqg2016.db
|
||||
npm run dev # Vite dev server
|
||||
npm run build # Production bundle → dist/
|
||||
npm run lint # ESLint
|
||||
# Build the SQLite database from source Excel files (requires Rust stable)
|
||||
pnpm run build:db
|
||||
|
||||
# Or build in two steps:
|
||||
pnpm run build:rust # compile the xlsxread binary
|
||||
./tools/xlsxread/target/release/xlsxread build \
|
||||
--schema tools/xlsxread/configs/thptqg2016-data.toml \
|
||||
--input data \
|
||||
--output public/thptqg2016.db
|
||||
|
||||
pnpm run dev # Vite dev server
|
||||
pnpm run build # Production bundle → dist/
|
||||
pnpm run lint # ESLint
|
||||
```
|
||||
|
||||
The GitHub Actions workflow (`.github/workflows/deploy.yml`) builds the DB, gzips it, and deploys to GitHub Pages on every push to `main`.
|
||||
The database is built by the `xlsxread` Rust binary (`tools/xlsxread/`), which
|
||||
reads the 119 mixed `.xls`/`.xlsx` files and auto-detects the column layout per
|
||||
file (`separate-scores`, `mapped`, or positional default). No Node.js Excel
|
||||
library is required at build time.
|
||||
|
||||
The GitHub Actions workflow (`.github/workflows/deploy.yml`) compiles
|
||||
`xlsxread`, builds the DB, gzips it, and deploys to GitHub Pages on every push
|
||||
to `main`.
|
||||
|
||||
## Tech stack
|
||||
|
||||
React 19 · Vite · sql.js (WASM) · better-sqlite3 (build-time only) · GitHub Pages
|
||||
React 19 · Vite · sql.js (WASM) · xlsxread (Rust, build-time) · GitHub Pages
|
||||
|
||||
## Documentation
|
||||
|
||||
|
||||
+3
-4
@@ -6,7 +6,8 @@
|
||||
"type": "module",
|
||||
"description": "Tra cứu điểm thi THPT QG 2016 - Hỗ trợ truy vấn SQL tùy chỉnh",
|
||||
"scripts": {
|
||||
"build:db": "node scripts/build-database.js",
|
||||
"build:rust": "cargo build --release --manifest-path tools/xlsxread/Cargo.toml",
|
||||
"build:db": "cargo build --release --manifest-path tools/xlsxread/Cargo.toml && ./tools/xlsxread/target/release/xlsxread build --schema tools/xlsxread/configs/thptqg2016-data.toml --input data --output public/thptqg2016.db",
|
||||
"dev": "vite",
|
||||
"build": "vite build",
|
||||
"preview": "vite preview",
|
||||
@@ -27,12 +28,10 @@
|
||||
"@types/react": "^19.2.14",
|
||||
"@types/react-dom": "^19.2.3",
|
||||
"@vitejs/plugin-react": "^6.0.1",
|
||||
"better-sqlite3": "^12.8.0",
|
||||
"eslint": "^9.39.4",
|
||||
"eslint-plugin-react-hooks": "^7.0.1",
|
||||
"eslint-plugin-react-refresh": "^0.5.2",
|
||||
"globals": "^17.4.0",
|
||||
"vite": "^8.0.4",
|
||||
"xlsx": "^0.18.5"
|
||||
"vite": "^8.0.4"
|
||||
}
|
||||
}
|
||||
Generated
-30
@@ -30,9 +30,6 @@ importers:
|
||||
'@vitejs/plugin-react':
|
||||
specifier: ^6.0.1
|
||||
version: 6.0.1(vite@8.0.12)
|
||||
better-sqlite3:
|
||||
specifier: ^12.8.0
|
||||
version: 12.9.0
|
||||
eslint:
|
||||
specifier: ^9.39.4
|
||||
version: 9.39.4
|
||||
@@ -48,9 +45,6 @@ importers:
|
||||
vite:
|
||||
specifier: ^8.0.4
|
||||
version: 8.0.12
|
||||
xlsx:
|
||||
specifier: ^0.18.5
|
||||
version: 0.18.5
|
||||
|
||||
packages:
|
||||
|
||||
@@ -379,10 +373,6 @@ packages:
|
||||
engines: {node: '>=6.0.0'}
|
||||
hasBin: true
|
||||
|
||||
better-sqlite3@12.9.0:
|
||||
resolution: {integrity: sha512-wqUv4Gm3toFpHDQmaKD4QhZm3g1DjUBI0yzS4UBl6lElUmXFYdTQmmEDpAFa5o8FiFiymURypEnfVHzILKaxqQ==}
|
||||
engines: {node: 20.x || 22.x || 23.x || 24.x || 25.x}
|
||||
|
||||
bindings@1.5.0:
|
||||
resolution: {integrity: sha512-p2q/t/mhvuOj/UeLlV6566GD/guowlr0hHxClI0W9m7MWYkL1F0hLo+0Aexs9HSPCtR1SXQ0TD3MMKrXZajbiQ==}
|
||||
|
||||
@@ -1034,11 +1024,6 @@ packages:
|
||||
wrappy@1.0.2:
|
||||
resolution: {integrity: sha512-l4Sp/DRseor9wL6EvV2+TuQn63dMkPjZ/sp9XkghTEbV9KlPS1xUsZ3u7/IQO4wxtcFB4bgpQPRcR3QCvezPcQ==}
|
||||
|
||||
xlsx@0.18.5:
|
||||
resolution: {integrity: sha512-dmg3LCjBPHZnQp5/F/+nnTa+miPJxUXB6vtk42YjBBKayDNagxGEeIdWApkYPOf3Z3pm3k62Knjzp7lMeTEtFQ==}
|
||||
engines: {node: '>=0.8'}
|
||||
hasBin: true
|
||||
|
||||
yallist@3.1.1:
|
||||
resolution: {integrity: sha512-a4UGQaWPH59mOXUYnAG2ewncQS4i4F43Tv3JoAM+s2VDAmS9NsK8GpDMLrCHPksFT7h3K6TOoUNn2pb7RoXx4g==}
|
||||
|
||||
@@ -1365,11 +1350,6 @@ snapshots:
|
||||
|
||||
baseline-browser-mapping@2.10.29: {}
|
||||
|
||||
better-sqlite3@12.9.0:
|
||||
dependencies:
|
||||
bindings: 1.5.0
|
||||
prebuild-install: 7.1.3
|
||||
|
||||
bindings@1.5.0:
|
||||
dependencies:
|
||||
file-uri-to-path: 1.0.0
|
||||
@@ -1942,16 +1922,6 @@ snapshots:
|
||||
|
||||
wrappy@1.0.2: {}
|
||||
|
||||
xlsx@0.18.5:
|
||||
dependencies:
|
||||
adler-32: 1.3.1
|
||||
cfb: 1.2.2
|
||||
codepage: 1.15.0
|
||||
crc-32: 1.2.2
|
||||
ssf: 0.11.2
|
||||
wmf: 1.0.2
|
||||
word: 0.3.0
|
||||
|
||||
yallist@3.1.1: {}
|
||||
|
||||
yocto-queue@0.1.0: {}
|
||||
|
||||
@@ -1,271 +0,0 @@
|
||||
import XLSX from "xlsx";
|
||||
import Database from "better-sqlite3";
|
||||
import fs from "fs";
|
||||
import path from "path";
|
||||
import { fileURLToPath } from "url";
|
||||
|
||||
const __dirname = path.dirname(fileURLToPath(import.meta.url));
|
||||
const DATA_DIR = path.join(__dirname, "..", "data");
|
||||
const DB_PATH = path.join(__dirname, "..", "public", "thptqg2016.db");
|
||||
|
||||
// Score patterns for the DIEM_THI string format
|
||||
const SCORE_PATTERNS = {
|
||||
toan: /Toán:\s*([\d.]+)/,
|
||||
ngu_van: /Ngữ văn:\s*([\d.]+)/,
|
||||
vat_ly: /Vật lí:\s*([\d.]+)/,
|
||||
hoa_hoc: /Hóa học:\s*([\d.]+)/,
|
||||
sinh_hoc: /Sinh học:\s*([\d.]+)/,
|
||||
lich_su: /Lịch sử:\s*([\d.]+)/,
|
||||
dia_ly: /Địa lí:\s*([\d.]+)/,
|
||||
tieng_anh: /Tiếng Anh:\s*([\d.]+)/,
|
||||
tieng_phap: /Tiếng Pháp:\s*([\d.]+)/,
|
||||
tieng_duc: /Tiếng Đức:\s*([\d.]+)/,
|
||||
tieng_nhat: /Tiếng Nhật:\s*([\d.]+)/,
|
||||
tieng_trung: /Tiếng Trung:\s*([\d.]+)/,
|
||||
};
|
||||
|
||||
const ALL_SCORE_FIELDS = Object.keys(SCORE_PATTERNS);
|
||||
|
||||
// Strip Vietnamese diacritics: "NGUYỄN BŨU LỘC" → "nguyen buu loc"
|
||||
function toAscii(str) {
|
||||
return str
|
||||
.normalize("NFD")
|
||||
.replace(/[\u0300-\u036f]/g, "")
|
||||
.replace(/đ/g, "d")
|
||||
.replace(/Đ/g, "D")
|
||||
.toLowerCase();
|
||||
}
|
||||
|
||||
// Parse score text "Toán: 3.75 Ngữ văn: 5.00 ..." into { toan: 3.75, ... }
|
||||
function parseScoreString(diemThi) {
|
||||
const scores = {};
|
||||
for (const [field, pattern] of Object.entries(SCORE_PATTERNS)) {
|
||||
const match = diemThi.match(pattern);
|
||||
if (match) scores[field] = parseFloat(match[1]);
|
||||
}
|
||||
return scores;
|
||||
}
|
||||
|
||||
// Detect header row by checking for known column names
|
||||
const KNOWN_HEADERS = new Set([
|
||||
"SOBAODANH", "SBD", "HO_TEN", "HOTEN", "HỌ TÊN",
|
||||
"NGAY_SINH", "TEN_CUMTHI", "GIOI_TINH", "DIEM_THI", "STT",
|
||||
"TOAN", "VAN", "LY", "HOA", "SINH ", "SU", "DIA",
|
||||
]);
|
||||
|
||||
function isHeaderRow(row) {
|
||||
if (!row || row.length < 2) return false;
|
||||
const first = String(row[0] || "").trim().toUpperCase();
|
||||
return KNOWN_HEADERS.has(first);
|
||||
}
|
||||
|
||||
// Detect which format a file uses based on its header row
|
||||
function detectFormat(headerRow) {
|
||||
if (!headerRow) return null;
|
||||
const cols = headerRow.map((c) => String(c || "").trim().toUpperCase());
|
||||
|
||||
// Format: SBD, HOTEN, TOAN, VAN, LY, HOA, SINH, SU, DIA, ...
|
||||
if (cols[0] === "SBD" && cols[2] === "TOAN") return "separate-scores";
|
||||
|
||||
// Build a column index map for flexible column ordering
|
||||
const map = {};
|
||||
for (let i = 0; i < cols.length; i++) {
|
||||
const c = cols[i];
|
||||
if (c === "SOBAODANH" || c === "SBD") map.sbd = i;
|
||||
else if (c === "HO_TEN" || c === "HOTEN" || c === "HỌ TÊN") map.ho_ten = i;
|
||||
else if (c === "NGAY_SINH") map.ngay_sinh = i;
|
||||
else if (c === "TEN_CUMTHI") map.ten_cum_thi = i;
|
||||
else if (c === "GIOI_TINH") map.gioi_tinh = i;
|
||||
else if (c === "DIEM_THI") map.diem_thi = i;
|
||||
}
|
||||
|
||||
if (map.sbd !== undefined && map.diem_thi !== undefined) {
|
||||
return { type: "mapped", map };
|
||||
}
|
||||
|
||||
return null;
|
||||
}
|
||||
|
||||
// Process a file with separate score columns (dhhanghai format)
|
||||
function processSeparateScoresRow(row) {
|
||||
const sbd = String(row[0] || "").trim();
|
||||
const hoTen = String(row[1] || "").trim();
|
||||
if (!sbd || !hoTen) return null;
|
||||
|
||||
return {
|
||||
so_bao_danh: sbd,
|
||||
ho_ten: hoTen,
|
||||
ho_ten_ascii: toAscii(hoTen),
|
||||
ngay_sinh: null,
|
||||
ten_cum_thi: null,
|
||||
gioi_tinh: null,
|
||||
toan: parseFloat(row[2]) || null,
|
||||
ngu_van: parseFloat(row[3]) || null,
|
||||
vat_ly: parseFloat(row[4]) || null,
|
||||
hoa_hoc: parseFloat(row[5]) || null,
|
||||
sinh_hoc: parseFloat(row[6]) || null,
|
||||
lich_su: parseFloat(row[7]) || null,
|
||||
dia_ly: parseFloat(row[8]) || null,
|
||||
// row[9]=NGOAINGUTN, row[10]=NGOAINGUTL, row[11]=NGOAINGU (total)
|
||||
tieng_anh: parseFloat(row[11]) || null,
|
||||
tieng_phap: null,
|
||||
tieng_duc: null,
|
||||
tieng_nhat: null,
|
||||
tieng_trung: null,
|
||||
};
|
||||
}
|
||||
|
||||
// Process a row using the column map
|
||||
function processMappedRow(row, map) {
|
||||
const sbd = String(row[map.sbd] || "").trim();
|
||||
const hoTen = String(row[map.ho_ten] || "").trim();
|
||||
if (!sbd || !hoTen) return null;
|
||||
|
||||
// Skip leaked header rows
|
||||
const sbdUpper = sbd.toUpperCase();
|
||||
if (KNOWN_HEADERS.has(sbdUpper) || KNOWN_HEADERS.has(hoTen.toUpperCase())) return null;
|
||||
|
||||
const ngaySinh = map.ngay_sinh !== undefined ? String(row[map.ngay_sinh] || "").trim() : null;
|
||||
const tenCumThi = map.ten_cum_thi !== undefined ? String(row[map.ten_cum_thi] || "").trim() : null;
|
||||
const rawGioiTinh = map.gioi_tinh !== undefined ? String(row[map.gioi_tinh] || "").trim() : null;
|
||||
// Normalize gender: only accept "Nam" or "Nữ"
|
||||
const gioiTinh = (rawGioiTinh === "Nam" || rawGioiTinh === "Nữ") ? rawGioiTinh : null;
|
||||
const diemThi = map.diem_thi !== undefined ? String(row[map.diem_thi] || "") : "";
|
||||
|
||||
const scores = parseScoreString(diemThi);
|
||||
|
||||
return {
|
||||
so_bao_danh: sbd,
|
||||
ho_ten: hoTen,
|
||||
ho_ten_ascii: toAscii(hoTen),
|
||||
ngay_sinh: ngaySinh || null,
|
||||
ten_cum_thi: tenCumThi || null,
|
||||
gioi_tinh: gioiTinh || null,
|
||||
...Object.fromEntries(ALL_SCORE_FIELDS.map((f) => [f, scores[f] ?? null])),
|
||||
};
|
||||
}
|
||||
|
||||
// Standard 6-column format without header: SBD, HO_TEN, NGAY_SINH, TEN_CUMTHI, GIOI_TINH, DIEM_THI
|
||||
const DEFAULT_MAP = {
|
||||
sbd: 0, ho_ten: 1, ngay_sinh: 2, ten_cum_thi: 3, gioi_tinh: 4, diem_thi: 5,
|
||||
};
|
||||
|
||||
function main() {
|
||||
fs.mkdirSync(path.dirname(DB_PATH), { recursive: true });
|
||||
if (fs.existsSync(DB_PATH)) fs.unlinkSync(DB_PATH);
|
||||
|
||||
const db = new Database(DB_PATH);
|
||||
|
||||
db.exec(`
|
||||
CREATE TABLE student (
|
||||
so_bao_danh TEXT PRIMARY KEY,
|
||||
ho_ten TEXT NOT NULL,
|
||||
ho_ten_ascii TEXT NOT NULL,
|
||||
ngay_sinh TEXT,
|
||||
ten_cum_thi TEXT,
|
||||
gioi_tinh TEXT,
|
||||
toan REAL,
|
||||
ngu_van REAL,
|
||||
vat_ly REAL,
|
||||
hoa_hoc REAL,
|
||||
sinh_hoc REAL,
|
||||
lich_su REAL,
|
||||
dia_ly REAL,
|
||||
tieng_anh REAL,
|
||||
tieng_phap REAL,
|
||||
tieng_duc REAL,
|
||||
tieng_nhat REAL,
|
||||
tieng_trung REAL
|
||||
);
|
||||
CREATE INDEX idx_ho_ten ON student(ho_ten);
|
||||
CREATE INDEX idx_ho_ten_ascii ON student(ho_ten_ascii);
|
||||
CREATE INDEX idx_ten_cum_thi ON student(ten_cum_thi);
|
||||
`);
|
||||
|
||||
const insert = db.prepare(`
|
||||
INSERT OR REPLACE INTO student
|
||||
(so_bao_danh, ho_ten, ho_ten_ascii, ngay_sinh, ten_cum_thi, gioi_tinh,
|
||||
toan, ngu_van, vat_ly, hoa_hoc, sinh_hoc, lich_su, dia_ly,
|
||||
tieng_anh, tieng_phap, tieng_duc, tieng_nhat, tieng_trung)
|
||||
VALUES
|
||||
(@so_bao_danh, @ho_ten, @ho_ten_ascii, @ngay_sinh, @ten_cum_thi, @gioi_tinh,
|
||||
@toan, @ngu_van, @vat_ly, @hoa_hoc, @sinh_hoc, @lich_su, @dia_ly,
|
||||
@tieng_anh, @tieng_phap, @tieng_duc, @tieng_nhat, @tieng_trung)
|
||||
`);
|
||||
|
||||
// Collect all Excel files (.xlsx and .xls)
|
||||
const files = fs.readdirSync(DATA_DIR)
|
||||
.filter((f) => f.endsWith(".xlsx") || f.endsWith(".xls"))
|
||||
.map((f) => path.join(DATA_DIR, f));
|
||||
|
||||
let totalRows = 0;
|
||||
let errorCount = 0;
|
||||
|
||||
const insertAll = db.transaction((files) => {
|
||||
for (const file of files) {
|
||||
const basename = path.basename(file);
|
||||
let fileRows = 0;
|
||||
|
||||
try {
|
||||
const wb = XLSX.readFile(file);
|
||||
const ws = wb.Sheets[wb.SheetNames[0]];
|
||||
const rows = XLSX.utils.sheet_to_json(ws, { header: 1 });
|
||||
if (rows.length === 0) continue;
|
||||
|
||||
let startRow = 0;
|
||||
let format = null;
|
||||
|
||||
if (isHeaderRow(rows[0])) {
|
||||
format = detectFormat(rows[0]);
|
||||
startRow = 1;
|
||||
}
|
||||
|
||||
for (let i = startRow; i < rows.length; i++) {
|
||||
const row = rows[i];
|
||||
if (!row || row.length < 2) continue;
|
||||
|
||||
try {
|
||||
let record;
|
||||
|
||||
if (format === "separate-scores") {
|
||||
record = processSeparateScoresRow(row);
|
||||
} else if (format && format.type === "mapped") {
|
||||
record = processMappedRow(row, format.map);
|
||||
} else {
|
||||
// No header or unrecognized: assume standard 6-column order
|
||||
record = processMappedRow(row, DEFAULT_MAP);
|
||||
}
|
||||
|
||||
if (!record) continue;
|
||||
insert.run(record);
|
||||
fileRows++;
|
||||
} catch {
|
||||
errorCount++;
|
||||
}
|
||||
}
|
||||
} catch (err) {
|
||||
console.error(`Failed to read ${basename}: ${err.message}`);
|
||||
}
|
||||
|
||||
totalRows += fileRows;
|
||||
console.log(` ${basename}: ${fileRows} rows`);
|
||||
}
|
||||
});
|
||||
|
||||
console.log(`Processing ${files.length} Excel files...\n`);
|
||||
insertAll(files);
|
||||
|
||||
db.exec("VACUUM");
|
||||
|
||||
const count = db.prepare("SELECT COUNT(*) as cnt FROM student").get();
|
||||
console.log(`\nDone! ${count.cnt} students in database.`);
|
||||
console.log(`Errors skipped: ${errorCount}`);
|
||||
console.log(`Output: ${DB_PATH}`);
|
||||
|
||||
const stat = fs.statSync(DB_PATH);
|
||||
console.log(`Size: ${(stat.size / 1024 / 1024).toFixed(1)} MB`);
|
||||
|
||||
db.close();
|
||||
}
|
||||
|
||||
main();
|
||||
Executable
+44
@@ -0,0 +1,44 @@
|
||||
#!/usr/bin/env bash
|
||||
# Re-sync xlsxread Rust source from thptqg2017.
|
||||
#
|
||||
# Source SHA: 8b4a755c115595bf1b937d749eb1133efb3a6e22 (chore/xlsxread-rust)
|
||||
#
|
||||
# Re-run this when xlsxread is updated upstream. The configs/ directory is
|
||||
# NOT synced — it contains thptqg2016-specific configs and test stubs that
|
||||
# must be maintained here independently.
|
||||
#
|
||||
# Usage:
|
||||
# ./tools/sync-from-thptqg2017.sh /path/to/thptqg2017
|
||||
#
|
||||
set -euo pipefail
|
||||
|
||||
SRC="${1:?Usage: $0 /path/to/thptqg2017}"
|
||||
|
||||
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||
DEST="$SCRIPT_DIR/xlsxread"
|
||||
|
||||
if [ ! -d "$SRC/tools/xlsxread" ]; then
|
||||
echo "Error: $SRC/tools/xlsxread not found" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
echo "Syncing from: $SRC/tools/xlsxread"
|
||||
echo "Syncing to: $DEST"
|
||||
echo ""
|
||||
|
||||
# Sync source, tests, and Cargo manifests — exclude build artifacts and dataset configs
|
||||
for item in src tests Cargo.toml Cargo.lock; do
|
||||
if [ -e "$SRC/tools/xlsxread/$item" ]; then
|
||||
cp -r "$SRC/tools/xlsxread/$item" "$DEST/"
|
||||
echo " synced: $item"
|
||||
fi
|
||||
done
|
||||
|
||||
echo ""
|
||||
echo "Sync complete."
|
||||
echo "Next steps:"
|
||||
echo " 1. Review src/ for breaking changes to config.rs / writer.rs that"
|
||||
echo " may affect format_detect_2016.rs or the thptqg2016-data.toml config."
|
||||
echo " 2. Rebuild: cargo build --release --manifest-path $DEST/Cargo.toml"
|
||||
echo " 3. Test: cargo test --manifest-path $DEST/Cargo.toml"
|
||||
echo " 4. Commit on a chore/xlsxread-sync-... branch."
|
||||
Generated
+1159
File diff suppressed because it is too large.
Load diff
@@ -0,0 +1,30 @@
|
||||
[package]
|
||||
name = "xlsxread"
|
||||
version = "0.1.0"
|
||||
edition = "2021"
|
||||
description = "Rust CLI replacing SheetJS xlsx build scripts for thptqg2017/thptqg2016"
|
||||
|
||||
[dependencies]
|
||||
calamine = "0.26"
|
||||
rusqlite = { version = "0.32", features = ["bundled"] }
|
||||
clap = { version = "4", features = ["derive"] }
|
||||
serde = { version = "1", features = ["derive"] }
|
||||
toml = "0.8"
|
||||
regex = "1"
|
||||
unicode-normalization = "0.1"
|
||||
thiserror = "1"
|
||||
anyhow = "1"
|
||||
glob = "0.3"
|
||||
|
||||
# zip is already a transitive dep of calamine; pin explicitly so tests can use it
|
||||
[dev-dependencies]
|
||||
zip = "2"
|
||||
rusqlite = { version = "0.32", features = ["bundled"] }
|
||||
|
||||
[[bin]]
|
||||
name = "xlsxread"
|
||||
path = "src/main.rs"
|
||||
|
||||
[[test]]
|
||||
name = "golden"
|
||||
path = "tests/golden.rs"
|
||||
@@ -0,0 +1,86 @@
|
||||
# Config for thptqg2016 data/ — 4 .xls + 115 .xlsx mixed files.
|
||||
#
|
||||
# Three column layouts exist across the 119 files; the binary selects the
|
||||
# right one per-file at runtime via format_detection = "thptqg2016":
|
||||
#
|
||||
# separate-scores header SBD(0)/HOTEN(1)/TOAN(2)... — dhhanghai files
|
||||
# mapped header SOBAODANH|SBD + DIEM_THI — most provinces
|
||||
# default no header; positional 6-col layout — remaining files
|
||||
#
|
||||
# Schema differences from thptqg2017:
|
||||
# + ten_cum_thi TEXT (exam-cluster name, TEN_CUMTHI column)
|
||||
# + gioi_tinh TEXT (gender: "Nam"/"Nữ", GIOI_TINH column)
|
||||
# + tieng_duc REAL (German)
|
||||
# + tieng_nhat REAL (Japanese)
|
||||
# - khtn, khxh, tieng_nga (not in 2016 dataset)
|
||||
#
|
||||
# sheet_mode = "all": several provinces overflow into Sheet2 (65k Excel row cap).
|
||||
# strip_blank_rows = false: no blank-row anomaly observed in this dataset.
|
||||
|
||||
format_detection = "thptqg2016"
|
||||
|
||||
[reader]
|
||||
sheet_mode = "all"
|
||||
strip_blank_rows = false
|
||||
|
||||
[validation]
|
||||
require_numeric_sbd = false
|
||||
require_nonempty_name = true
|
||||
require_nonempty_sbd = true
|
||||
|
||||
[header]
|
||||
# Tokens that identify a header row by first-cell content (uppercased).
|
||||
# Covers both SOBAODANH-style and SBD-style headers.
|
||||
tokens = ["SOBAODANH", "SBD", "HO_TEN", "HOTEN", "HỌ TÊN", "STT"]
|
||||
|
||||
[schema]
|
||||
ddl = """
|
||||
CREATE TABLE student (
|
||||
so_bao_danh TEXT PRIMARY KEY,
|
||||
ho_ten TEXT NOT NULL,
|
||||
ho_ten_ascii TEXT NOT NULL,
|
||||
ngay_sinh TEXT,
|
||||
ten_cum_thi TEXT,
|
||||
gioi_tinh TEXT,
|
||||
toan REAL,
|
||||
ngu_van REAL,
|
||||
vat_ly REAL,
|
||||
hoa_hoc REAL,
|
||||
sinh_hoc REAL,
|
||||
lich_su REAL,
|
||||
dia_ly REAL,
|
||||
tieng_anh REAL,
|
||||
tieng_phap REAL,
|
||||
tieng_duc REAL,
|
||||
tieng_nhat REAL,
|
||||
tieng_trung REAL
|
||||
);
|
||||
CREATE INDEX idx_ho_ten ON student(ho_ten);
|
||||
CREATE INDEX idx_ho_ten_ascii ON student(ho_ten_ascii);
|
||||
CREATE INDEX idx_ten_cum_thi ON student(ten_cum_thi);
|
||||
"""
|
||||
|
||||
[scores]
|
||||
toan = 'Toán:\s*(\d+(?:\.\d+)?)'
|
||||
ngu_van = 'Ngữ văn:\s*(\d+(?:\.\d+)?)'
|
||||
vat_ly = 'Vật lí:\s*(\d+(?:\.\d+)?)'
|
||||
hoa_hoc = 'Hóa học:\s*(\d+(?:\.\d+)?)'
|
||||
sinh_hoc = 'Sinh học:\s*(\d+(?:\.\d+)?)'
|
||||
lich_su = 'Lịch sử:\s*(\d+(?:\.\d+)?)'
|
||||
dia_ly = 'Địa lí:\s*(\d+(?:\.\d+)?)'
|
||||
tieng_anh = 'Tiếng Anh:\s*(\d+(?:\.\d+)?)'
|
||||
tieng_phap = 'Tiếng Pháp:\s*(\d+(?:\.\d+)?)'
|
||||
tieng_duc = 'Tiếng Đức:\s*(\d+(?:\.\d+)?)'
|
||||
tieng_nhat = 'Tiếng Nhật:\s*(\d+(?:\.\d+)?)'
|
||||
tieng_trung = 'Tiếng Trung:\s*(\d+(?:\.\d+)?)'
|
||||
|
||||
[insert]
|
||||
sql = """
|
||||
INSERT OR REPLACE INTO student
|
||||
(so_bao_danh, ho_ten, ho_ten_ascii, ngay_sinh, ten_cum_thi, gioi_tinh,
|
||||
toan, ngu_van, vat_ly, hoa_hoc, sinh_hoc,
|
||||
lich_su, dia_ly,
|
||||
tieng_anh, tieng_phap, tieng_duc, tieng_nhat, tieng_trung)
|
||||
VALUES
|
||||
(?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
|
||||
"""
|
||||
@@ -0,0 +1,76 @@
|
||||
# Test-only config used by golden integration tests (data-old variant).
|
||||
# Same column layout as thptqg2017-data.toml but with:
|
||||
# - sheet_mode = "first" (reads only first sheet)
|
||||
# - require_numeric_sbd = true (rejects non-digit SBDs)
|
||||
# Schema and INSERT match SCORE_FIELDS_2017 (14 score cols).
|
||||
|
||||
[reader]
|
||||
sheet_mode = "first"
|
||||
strip_blank_rows = false
|
||||
|
||||
[columns]
|
||||
ho_ten = 0
|
||||
ngay_sinh = 1
|
||||
so_bao_danh = 2
|
||||
diem_thi = 3
|
||||
|
||||
[validation]
|
||||
require_numeric_sbd = true
|
||||
require_nonempty_name = true
|
||||
require_nonempty_sbd = true
|
||||
|
||||
[header]
|
||||
tokens = ["HO_TEN", "HỌ TÊN", "STT"]
|
||||
|
||||
[schema]
|
||||
ddl = """
|
||||
CREATE TABLE student (
|
||||
so_bao_danh TEXT PRIMARY KEY,
|
||||
ho_ten TEXT NOT NULL,
|
||||
ho_ten_ascii TEXT NOT NULL,
|
||||
ngay_sinh TEXT,
|
||||
toan REAL,
|
||||
ngu_van REAL,
|
||||
vat_ly REAL,
|
||||
hoa_hoc REAL,
|
||||
sinh_hoc REAL,
|
||||
khtn REAL,
|
||||
lich_su REAL,
|
||||
dia_ly REAL,
|
||||
gdcd REAL,
|
||||
khxh REAL,
|
||||
tieng_anh REAL,
|
||||
tieng_phap REAL,
|
||||
tieng_nga REAL,
|
||||
tieng_trung REAL
|
||||
);
|
||||
CREATE INDEX idx_ho_ten ON student(ho_ten);
|
||||
CREATE INDEX idx_ho_ten_ascii ON student(ho_ten_ascii);
|
||||
"""
|
||||
|
||||
[scores]
|
||||
toan = 'Toán:\s*(\d+(?:\.\d+)?)'
|
||||
ngu_van = 'Ngữ văn:\s*(\d+(?:\.\d+)?)'
|
||||
vat_ly = 'Vật lí:\s*(\d+(?:\.\d+)?)'
|
||||
hoa_hoc = 'Hóa học:\s*(\d+(?:\.\d+)?)'
|
||||
sinh_hoc = 'Sinh học:\s*(\d+(?:\.\d+)?)'
|
||||
khtn = 'KHTN:\s*(\d+(?:\.\d+)?)'
|
||||
lich_su = 'Lịch sử:\s*(\d+(?:\.\d+)?)'
|
||||
dia_ly = 'Địa lí:\s*(\d+(?:\.\d+)?)'
|
||||
gdcd = 'GDCD:\s*(\d+(?:\.\d+)?)'
|
||||
khxh = 'KHXH:\s*(\d+(?:\.\d+)?)'
|
||||
tieng_anh = 'Tiếng Anh:\s*(\d+(?:\.\d+)?)'
|
||||
tieng_phap = 'Tiếng Pháp:\s*(\d+(?:\.\d+)?)'
|
||||
tieng_nga = 'Tiếng Nga:\s*(\d+(?:\.\d+)?)'
|
||||
tieng_trung = 'Tiếng Trung:\s*(\d+(?:\.\d+)?)'
|
||||
|
||||
[insert]
|
||||
sql = """
|
||||
INSERT OR REPLACE INTO student
|
||||
(so_bao_danh, ho_ten, ho_ten_ascii, ngay_sinh,
|
||||
toan, ngu_van, vat_ly, hoa_hoc, sinh_hoc, khtn,
|
||||
lich_su, dia_ly, gdcd, khxh,
|
||||
tieng_anh, tieng_phap, tieng_nga, tieng_trung)
|
||||
VALUES
|
||||
(?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
|
||||
"""
|
||||
@@ -0,0 +1,76 @@
|
||||
# Test-only config used by golden integration tests.
|
||||
# Matches the fixture row layout produced by tests/golden.rs: write_xlsx()
|
||||
# header: HO_TEN(0) NGAY_SINH(1) SO_BAO_DANH(2) DIEM_THI(3)
|
||||
# Schema and INSERT match SCORE_FIELDS_2017 (14 score cols) so run_build_cmd
|
||||
# in golden.rs can call insert_row(..., SCORE_FIELDS) without modification.
|
||||
|
||||
[reader]
|
||||
sheet_mode = "all"
|
||||
strip_blank_rows = false
|
||||
|
||||
[columns]
|
||||
ho_ten = 0
|
||||
ngay_sinh = 1
|
||||
so_bao_danh = 2
|
||||
diem_thi = 3
|
||||
|
||||
[validation]
|
||||
require_numeric_sbd = false
|
||||
require_nonempty_name = true
|
||||
require_nonempty_sbd = true
|
||||
|
||||
[header]
|
||||
tokens = ["HO_TEN", "HỌ TÊN", "STT"]
|
||||
|
||||
[schema]
|
||||
ddl = """
|
||||
CREATE TABLE student (
|
||||
so_bao_danh TEXT PRIMARY KEY,
|
||||
ho_ten TEXT NOT NULL,
|
||||
ho_ten_ascii TEXT NOT NULL,
|
||||
ngay_sinh TEXT,
|
||||
toan REAL,
|
||||
ngu_van REAL,
|
||||
vat_ly REAL,
|
||||
hoa_hoc REAL,
|
||||
sinh_hoc REAL,
|
||||
khtn REAL,
|
||||
lich_su REAL,
|
||||
dia_ly REAL,
|
||||
gdcd REAL,
|
||||
khxh REAL,
|
||||
tieng_anh REAL,
|
||||
tieng_phap REAL,
|
||||
tieng_nga REAL,
|
||||
tieng_trung REAL
|
||||
);
|
||||
CREATE INDEX idx_ho_ten ON student(ho_ten);
|
||||
CREATE INDEX idx_ho_ten_ascii ON student(ho_ten_ascii);
|
||||
"""
|
||||
|
||||
[scores]
|
||||
toan = 'Toán:\s*(\d+(?:\.\d+)?)'
|
||||
ngu_van = 'Ngữ văn:\s*(\d+(?:\.\d+)?)'
|
||||
vat_ly = 'Vật lí:\s*(\d+(?:\.\d+)?)'
|
||||
hoa_hoc = 'Hóa học:\s*(\d+(?:\.\d+)?)'
|
||||
sinh_hoc = 'Sinh học:\s*(\d+(?:\.\d+)?)'
|
||||
khtn = 'KHTN:\s*(\d+(?:\.\d+)?)'
|
||||
lich_su = 'Lịch sử:\s*(\d+(?:\.\d+)?)'
|
||||
dia_ly = 'Địa lí:\s*(\d+(?:\.\d+)?)'
|
||||
gdcd = 'GDCD:\s*(\d+(?:\.\d+)?)'
|
||||
khxh = 'KHXH:\s*(\d+(?:\.\d+)?)'
|
||||
tieng_anh = 'Tiếng Anh:\s*(\d+(?:\.\d+)?)'
|
||||
tieng_phap = 'Tiếng Pháp:\s*(\d+(?:\.\d+)?)'
|
||||
tieng_nga = 'Tiếng Nga:\s*(\d+(?:\.\d+)?)'
|
||||
tieng_trung = 'Tiếng Trung:\s*(\d+(?:\.\d+)?)'
|
||||
|
||||
[insert]
|
||||
sql = """
|
||||
INSERT OR REPLACE INTO student
|
||||
(so_bao_danh, ho_ten, ho_ten_ascii, ngay_sinh,
|
||||
toan, ngu_van, vat_ly, hoa_hoc, sinh_hoc, khtn,
|
||||
lich_su, dia_ly, gdcd, khxh,
|
||||
tieng_anh, tieng_phap, tieng_nga, tieng_trung)
|
||||
VALUES
|
||||
(?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
|
||||
"""
|
||||
@@ -0,0 +1,182 @@
|
||||
/// Audit subcommand: replicates audit-row-counts.js exactly.
|
||||
///
|
||||
/// Reads all .xlsx files from the input directory (sheet 0 only, matching the
|
||||
/// JS script's behaviour at audit-row-counts.js:33), collects distinct SBDs
|
||||
/// into a HashSet, then queries `SELECT COUNT(*) FROM student` from the DB.
|
||||
/// Prints the same lines as audit-row-counts.js:54-62 and exits 0 on match,
|
||||
/// 1 on mismatch.
|
||||
use std::collections::HashSet;
|
||||
use std::path::Path;
|
||||
|
||||
use calamine::{open_workbook_auto, Data, Reader};
|
||||
|
||||
use crate::config::DatasetConfig;
|
||||
use crate::error::BuildError;
|
||||
use crate::reader::is_header_row;
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Audit result
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
pub struct AuditResult {
|
||||
pub total_data_rows: u64,
|
||||
pub both_empty: u64,
|
||||
pub empty_name: u64,
|
||||
pub empty_sbd: u64,
|
||||
pub distinct_sbds: usize,
|
||||
pub db_count: i64,
|
||||
pub matched: bool,
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Main audit logic
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// Collect distinct SBDs from all xlsx files in `input_dir`, query `db_path`,
|
||||
/// print the audit report and return the result.
|
||||
///
|
||||
/// The JS script reads only sheet 0 for every file (audit-row-counts.js:33).
|
||||
/// Unlike build-database.js, the audit script does NOT iterate all sheets.
|
||||
pub fn run_audit(
|
||||
input_dir: &Path,
|
||||
db_path: &Path,
|
||||
cfg: &DatasetConfig,
|
||||
) -> Result<AuditResult, BuildError> {
|
||||
// Collect .xlsx files (audit-row-counts.js only checks .xlsx — line 15)
|
||||
let mut files: Vec<std::path::PathBuf> = std::fs::read_dir(input_dir)
|
||||
.map_err(|e| BuildError::Io {
|
||||
path: input_dir.display().to_string(),
|
||||
source: e,
|
||||
})?
|
||||
.filter_map(|e| e.ok())
|
||||
.map(|e| e.path())
|
||||
.filter(|p| {
|
||||
p.is_file()
|
||||
&& p.extension()
|
||||
.and_then(|e| e.to_str())
|
||||
.map(|e| e.eq_ignore_ascii_case("xlsx"))
|
||||
.unwrap_or(false)
|
||||
})
|
||||
.collect();
|
||||
files.sort();
|
||||
|
||||
let mut all_sbd: HashSet<String> = HashSet::new();
|
||||
let mut total_data_rows: u64 = 0;
|
||||
let mut empty_name: u64 = 0;
|
||||
let mut empty_sbd: u64 = 0;
|
||||
let mut both_empty: u64 = 0;
|
||||
|
||||
for file in &files {
|
||||
let path_str = file.display().to_string();
|
||||
let mut workbook = open_workbook_auto(file).map_err(|e| BuildError::Calamine {
|
||||
path: path_str.clone(),
|
||||
source: e,
|
||||
})?;
|
||||
|
||||
let sheet_names = workbook.sheet_names().to_vec();
|
||||
if sheet_names.is_empty() {
|
||||
continue;
|
||||
}
|
||||
|
||||
// audit-row-counts.js reads only sheet 0 (line 33: wb.SheetNames[0])
|
||||
let range =
|
||||
workbook
|
||||
.worksheet_range(&sheet_names[0])
|
||||
.map_err(|e| BuildError::Calamine {
|
||||
path: path_str.clone(),
|
||||
source: e,
|
||||
})?;
|
||||
|
||||
let mut first_row = true;
|
||||
for raw in range.rows() {
|
||||
let row: Vec<Data> = raw.to_vec();
|
||||
|
||||
// Skip header row on first row only
|
||||
if first_row {
|
||||
first_row = false;
|
||||
if is_header_row(&row, &cfg.header) {
|
||||
continue;
|
||||
}
|
||||
}
|
||||
|
||||
total_data_rows += 1;
|
||||
|
||||
// For format_detection configs the audit uses positional defaults (col 0 = SBD,
|
||||
// col 1 = HO_TEN) because the audit is best-effort and mirrors the JS script
|
||||
// which also uses a fixed column assumption (audit-row-counts.js:33–36).
|
||||
let (ho_ten_col, sbd_col) = cfg
|
||||
.columns
|
||||
.as_ref()
|
||||
.map(|c| (c.ho_ten, c.so_bao_danh))
|
||||
.unwrap_or((1, 0));
|
||||
let ho_ten = row
|
||||
.get(ho_ten_col)
|
||||
.map(|c| c.to_string().trim().to_owned())
|
||||
.unwrap_or_default();
|
||||
let sbd = row
|
||||
.get(sbd_col)
|
||||
.map(|c| c.to_string().trim().to_owned())
|
||||
.unwrap_or_default();
|
||||
|
||||
if ho_ten.is_empty() && sbd.is_empty() {
|
||||
both_empty += 1;
|
||||
continue;
|
||||
}
|
||||
if ho_ten.is_empty() {
|
||||
empty_name += 1;
|
||||
}
|
||||
if sbd.is_empty() {
|
||||
empty_sbd += 1;
|
||||
}
|
||||
if !sbd.is_empty() {
|
||||
all_sbd.insert(sbd);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Query DB count
|
||||
let conn =
|
||||
rusqlite::Connection::open_with_flags(db_path, rusqlite::OpenFlags::SQLITE_OPEN_READ_ONLY)?;
|
||||
let db_count: i64 = conn.query_row("SELECT COUNT(*) FROM student", [], |row| row.get(0))?;
|
||||
|
||||
let distinct_sbds = all_sbd.len();
|
||||
let matched = distinct_sbds as i64 == db_count;
|
||||
|
||||
Ok(AuditResult {
|
||||
total_data_rows,
|
||||
both_empty,
|
||||
empty_name,
|
||||
empty_sbd,
|
||||
distinct_sbds,
|
||||
db_count,
|
||||
matched,
|
||||
})
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Print audit report — mirrors audit-row-counts.js:54-62 exactly
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
pub fn print_audit_report(r: &AuditResult) {
|
||||
println!("=== Source vs DB ===");
|
||||
println!(
|
||||
"Source: total data rows across all files: {}",
|
||||
r.total_data_rows
|
||||
);
|
||||
println!(
|
||||
"Source: rows with empty name AND sbd (skipped): {}",
|
||||
r.both_empty
|
||||
);
|
||||
println!("Source: rows with missing name only: {}", r.empty_name);
|
||||
println!("Source: rows with missing sbd only: {}", r.empty_sbd);
|
||||
println!("Source: distinct SBDs: {}", r.distinct_sbds);
|
||||
println!("DB: row count: {}", r.db_count);
|
||||
println!(
|
||||
"Match: {}",
|
||||
if r.matched {
|
||||
"YES — all unique SBDs accounted for".to_string()
|
||||
} else {
|
||||
format!("NO — gap of {}", r.distinct_sbds as i64 - r.db_count)
|
||||
}
|
||||
);
|
||||
}
|
||||
@@ -0,0 +1,48 @@
|
||||
/// CLI argument structs via clap derive.
|
||||
use std::path::PathBuf;
|
||||
|
||||
use clap::{Parser, Subcommand};
|
||||
|
||||
#[derive(Parser)]
|
||||
#[command(
|
||||
name = "xlsxread",
|
||||
version,
|
||||
about = "Read .xls/.xlsx files and build SQLite databases for thptqg datasets"
|
||||
)]
|
||||
pub struct Cli {
|
||||
#[command(subcommand)]
|
||||
pub cmd: Cmd,
|
||||
}
|
||||
|
||||
#[derive(Subcommand)]
|
||||
pub enum Cmd {
|
||||
/// Read input spreadsheets and write a SQLite database
|
||||
Build {
|
||||
/// Path to the dataset TOML config file
|
||||
#[arg(long)]
|
||||
schema: PathBuf,
|
||||
|
||||
/// Directory containing the .xls / .xlsx source files
|
||||
#[arg(long)]
|
||||
input: PathBuf,
|
||||
|
||||
/// Output SQLite database path
|
||||
#[arg(long)]
|
||||
output: PathBuf,
|
||||
},
|
||||
|
||||
/// Audit: compare distinct SBD count from xlsx files vs DB row count
|
||||
Audit {
|
||||
/// Path to the dataset TOML config file
|
||||
#[arg(long)]
|
||||
schema: PathBuf,
|
||||
|
||||
/// Directory containing the .xlsx source files
|
||||
#[arg(long)]
|
||||
input: PathBuf,
|
||||
|
||||
/// SQLite database to compare against
|
||||
#[arg(long)]
|
||||
db: PathBuf,
|
||||
},
|
||||
}
|
||||
@@ -0,0 +1,188 @@
|
||||
use std::collections::HashMap;
|
||||
use std::fs;
|
||||
use std::path::Path;
|
||||
|
||||
use serde::Deserialize;
|
||||
|
||||
use crate::error::BuildError;
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Top-level dataset configuration loaded from a .toml file
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
#[derive(Debug, Deserialize, Clone)]
|
||||
pub struct DatasetConfig {
|
||||
pub reader: ReaderCfg,
|
||||
/// Fixed column indices. Optional when format_detection handles per-file mapping.
|
||||
#[serde(default)]
|
||||
pub columns: Option<ColumnMap>,
|
||||
pub validation: ValidationCfg,
|
||||
pub header: HeaderCfg,
|
||||
pub schema: SchemaCfg,
|
||||
/// field name → regex source string (one entry per scoreable subject)
|
||||
pub scores: HashMap<String, String>,
|
||||
pub insert: InsertCfg,
|
||||
/// When set to "thptqg2016", enables per-file format auto-detection.
|
||||
/// Each file's header row is inspected at runtime to choose the right
|
||||
/// column layout (separate-scores / mapped / default-positional).
|
||||
#[serde(default)]
|
||||
pub format_detection: Option<String>,
|
||||
}
|
||||
|
||||
#[derive(Debug, Deserialize, Clone)]
|
||||
pub struct ReaderCfg {
|
||||
/// "all" → iterate every sheet (handles HCM/HN overflow); "first" → sheet 0 only
|
||||
pub sheet_mode: SheetMode,
|
||||
/// If true, skip rows where every cell is empty/null before counting (data-old2 quirk)
|
||||
pub strip_blank_rows: bool,
|
||||
}
|
||||
|
||||
#[derive(Debug, Deserialize, Clone, PartialEq, Eq)]
|
||||
#[serde(rename_all = "lowercase")]
|
||||
pub enum SheetMode {
|
||||
All,
|
||||
First,
|
||||
}
|
||||
|
||||
/// Zero-indexed column positions in the source spreadsheet row.
|
||||
/// Used by thptqg2017 configs. thptqg2016 uses runtime format detection instead.
|
||||
#[derive(Debug, Deserialize, Clone)]
|
||||
pub struct ColumnMap {
|
||||
pub ho_ten: usize,
|
||||
pub ngay_sinh: usize,
|
||||
pub so_bao_danh: usize,
|
||||
pub diem_thi: usize,
|
||||
}
|
||||
|
||||
#[derive(Debug, Deserialize, Clone)]
|
||||
pub struct ValidationCfg {
|
||||
/// build-database-old.js / -old2.js require soBaoDanh to match ^\d+$
|
||||
pub require_numeric_sbd: bool,
|
||||
pub require_nonempty_name: bool,
|
||||
pub require_nonempty_sbd: bool,
|
||||
}
|
||||
|
||||
#[derive(Debug, Deserialize, Clone)]
|
||||
pub struct HeaderCfg {
|
||||
/// Tokens to match against row[0].to_uppercase() to detect a header row
|
||||
pub tokens: Vec<String>,
|
||||
}
|
||||
|
||||
#[derive(Debug, Deserialize, Clone)]
|
||||
pub struct SchemaCfg {
|
||||
/// DDL executed verbatim before inserts (CREATE TABLE + CREATE INDEX)
|
||||
pub ddl: String,
|
||||
}
|
||||
|
||||
#[derive(Debug, Deserialize, Clone)]
|
||||
pub struct InsertCfg {
|
||||
/// Parameterised INSERT OR REPLACE SQL using :named_param style
|
||||
pub sql: String,
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Loader
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
pub fn load_config(path: &Path) -> Result<DatasetConfig, BuildError> {
|
||||
let text = fs::read_to_string(path).map_err(|e| BuildError::Io {
|
||||
path: path.display().to_string(),
|
||||
source: e,
|
||||
})?;
|
||||
let cfg: DatasetConfig = toml::from_str(&text)?;
|
||||
Ok(cfg)
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Unit tests
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
const SAMPLE_TOML: &str = r#"
|
||||
[reader]
|
||||
sheet_mode = "all"
|
||||
strip_blank_rows = false
|
||||
|
||||
[columns]
|
||||
ho_ten = 0
|
||||
ngay_sinh = 1
|
||||
so_bao_danh = 2
|
||||
diem_thi = 3
|
||||
|
||||
[validation]
|
||||
require_numeric_sbd = false
|
||||
require_nonempty_name = true
|
||||
require_nonempty_sbd = true
|
||||
|
||||
[header]
|
||||
tokens = ["HO_TEN", "HỌ TÊN", "STT"]
|
||||
|
||||
[schema]
|
||||
ddl = "CREATE TABLE student (so_bao_danh TEXT PRIMARY KEY);"
|
||||
|
||||
[scores]
|
||||
toan = 'Toán:\s*(\d+(?:\.\d+)?)'
|
||||
ngu_van = 'Ngữ văn:\s*(\d+(?:\.\d+)?)'
|
||||
|
||||
[insert]
|
||||
sql = "INSERT OR REPLACE INTO student (so_bao_danh) VALUES (:so_bao_danh)"
|
||||
"#;
|
||||
|
||||
#[test]
|
||||
fn config_round_trip() {
|
||||
let cfg: DatasetConfig = toml::from_str(SAMPLE_TOML).expect("parse failed");
|
||||
assert_eq!(cfg.reader.sheet_mode, SheetMode::All);
|
||||
assert!(!cfg.reader.strip_blank_rows);
|
||||
let cols = cfg.columns.as_ref().unwrap();
|
||||
assert_eq!(cols.ho_ten, 0);
|
||||
assert_eq!(cols.diem_thi, 3);
|
||||
assert!(!cfg.validation.require_numeric_sbd);
|
||||
assert!(cfg.validation.require_nonempty_name);
|
||||
assert_eq!(cfg.header.tokens.len(), 3);
|
||||
assert!(cfg.scores.contains_key("toan"));
|
||||
assert!(cfg.scores.contains_key("ngu_van"));
|
||||
assert!(cfg.format_detection.is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn config_first_sheet_mode() {
|
||||
let toml_str = SAMPLE_TOML.replace(r#"sheet_mode = "all""#, r#"sheet_mode = "first""#);
|
||||
let cfg: DatasetConfig = toml::from_str(&toml_str).expect("parse failed");
|
||||
assert_eq!(cfg.reader.sheet_mode, SheetMode::First);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn config_format_detection_field() {
|
||||
// Configs without [columns] and with format_detection = "thptqg2016" parse correctly
|
||||
let toml_str = r#"
|
||||
format_detection = "thptqg2016"
|
||||
|
||||
[reader]
|
||||
sheet_mode = "all"
|
||||
strip_blank_rows = false
|
||||
|
||||
[validation]
|
||||
require_numeric_sbd = false
|
||||
require_nonempty_name = true
|
||||
require_nonempty_sbd = true
|
||||
|
||||
[header]
|
||||
tokens = ["SBD", "SOBAODANH", "STT"]
|
||||
|
||||
[schema]
|
||||
ddl = "CREATE TABLE student (so_bao_danh TEXT PRIMARY KEY);"
|
||||
|
||||
[scores]
|
||||
toan = 'Toán:\s*(\d+(?:\.\d+)?)'
|
||||
|
||||
[insert]
|
||||
sql = "INSERT OR REPLACE INTO student (so_bao_danh) VALUES (?)"
|
||||
"#;
|
||||
let cfg: DatasetConfig = toml::from_str(toml_str).expect("parse failed");
|
||||
assert_eq!(cfg.format_detection.as_deref(), Some("thptqg2016"));
|
||||
assert!(cfg.columns.is_none());
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,34 @@
|
||||
use thiserror::Error;
|
||||
|
||||
#[derive(Debug, Error)]
|
||||
pub enum BuildError {
|
||||
#[error("I/O error for {path}: {source}")]
|
||||
Io {
|
||||
path: String,
|
||||
#[source]
|
||||
source: std::io::Error,
|
||||
},
|
||||
|
||||
#[error("Calamine error for {path}: {source}")]
|
||||
Calamine {
|
||||
path: String,
|
||||
#[source]
|
||||
source: calamine::Error,
|
||||
},
|
||||
|
||||
#[error("SQLite error: {0}")]
|
||||
Sqlite(#[from] rusqlite::Error),
|
||||
|
||||
#[error("Config parse error: {0}")]
|
||||
Config(#[from] toml::de::Error),
|
||||
|
||||
#[error("Regex compile error for pattern '{pattern}': {source}")]
|
||||
Regex {
|
||||
pattern: String,
|
||||
#[source]
|
||||
source: regex::Error,
|
||||
},
|
||||
|
||||
#[error("Schema has no sheets in file: {0}")]
|
||||
NoSheets(String),
|
||||
}
|
||||
@@ -0,0 +1,549 @@
|
||||
/// Per-file format auto-detection for the thptqg2016 dataset.
|
||||
///
|
||||
/// Translates `detectFormat` from scripts/build-database.js (lines 63–87) and the
|
||||
/// three row-processing functions (lines 90–146) into Rust.
|
||||
///
|
||||
/// The JS source has three formats:
|
||||
///
|
||||
/// 1. `separate-scores` — header row[0]=="SBD" && row[2]=="TOAN"
|
||||
/// Columns: SBD(0) HOTEN(1) TOAN(2) VAN(3) LY(4) HOA(5) SINH(6) SU(7) DIA(8)
|
||||
/// NGOAINGUTN(9) NGOAINGUTL(10) NGOAINGU-total(11)
|
||||
/// → maps col 11 → tieng_anh; no ngay_sinh / ten_cum_thi / gioi_tinh / DIEM_THI
|
||||
/// → JS: build-database.js:90–116 (processSeparateScoresRow)
|
||||
///
|
||||
/// 2. `mapped` — header row has SOBAODANH|SBD and DIEM_THI columns
|
||||
/// → dynamic column indices built from header names
|
||||
/// → JS: build-database.js:119–146 (processMappedRow with map from detectFormat)
|
||||
///
|
||||
/// 3. `default` — no recognised header; positional 6-col layout
|
||||
/// SBD(0) HO_TEN(1) NGAY_SINH(2) TEN_CUMTHI(3) GIOI_TINH(4) DIEM_THI(5)
|
||||
/// → JS: build-database.js:149–151 DEFAULT_MAP + processMappedRow
|
||||
///
|
||||
/// JS citations are line numbers in /config/workspace/tiennm99/thptqg2016/scripts/build-database.js.
|
||||
use calamine::Data;
|
||||
|
||||
use crate::transform::{parse_scores, to_ascii, CompiledPatterns, ParsedRow};
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Known header tokens (mirrors JS KNOWN_HEADERS set, build-database.js:50–54)
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// Upper-cased strings that identify a header row's first cell.
|
||||
/// Mirrors `KNOWN_HEADERS` in build-database.js:50–54.
|
||||
const KNOWN_HEADERS: &[&str] = &[
|
||||
"SOBAODANH",
|
||||
"SBD",
|
||||
"HO_TEN",
|
||||
"HOTEN",
|
||||
"HỌ TÊN",
|
||||
"NGAY_SINH",
|
||||
"TEN_CUMTHI",
|
||||
"GIOI_TINH",
|
||||
"DIEM_THI",
|
||||
"STT",
|
||||
"TOAN",
|
||||
"VAN",
|
||||
"LY",
|
||||
"HOA",
|
||||
"SINH ",
|
||||
"SU",
|
||||
"DIA",
|
||||
];
|
||||
|
||||
/// Returns true when `row[0]` (uppercased, trimmed) is in the known-headers set.
|
||||
/// Mirrors `isHeaderRow` at build-database.js:56–60.
|
||||
pub fn is_header_row_2016(row: &[Data]) -> bool {
|
||||
if row.len() < 2 {
|
||||
return false;
|
||||
}
|
||||
let first = row[0].to_string().trim().to_uppercase();
|
||||
KNOWN_HEADERS.iter().any(|h| *h == first.as_str())
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Detected per-file format
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// The three layouts the thptqg2016 dataset uses, detected per file.
|
||||
#[derive(Debug, Clone)]
|
||||
pub enum DetectedFormat {
|
||||
/// SBD/HOTEN/TOAN/VAN/LY/HOA/SINH/SU/DIA/NGOAINGUTN/NGOAINGUTL/NGOAINGU columns.
|
||||
/// Corresponds to the dhhanghai-style files. build-database.js:67–68.
|
||||
SeparateScores,
|
||||
/// Header present with SOBAODANH|SBD and DIEM_THI; dynamic column indices.
|
||||
/// build-database.js:70–86.
|
||||
Mapped {
|
||||
sbd: usize,
|
||||
ho_ten: usize,
|
||||
ngay_sinh: Option<usize>,
|
||||
ten_cum_thi: Option<usize>,
|
||||
gioi_tinh: Option<usize>,
|
||||
diem_thi: usize,
|
||||
},
|
||||
/// No recognised header; standard 6-column positional layout.
|
||||
/// build-database.js:149–151 DEFAULT_MAP.
|
||||
Default,
|
||||
}
|
||||
|
||||
/// Inspect a header row and decide which format applies.
|
||||
/// Returns `None` when the header is present but unrecognised (treated as Default).
|
||||
///
|
||||
/// Mirrors `detectFormat` at build-database.js:63–87.
|
||||
pub fn detect_format(header_row: &[Data]) -> DetectedFormat {
|
||||
let cols: Vec<String> = header_row
|
||||
.iter()
|
||||
.map(|c| c.to_string().trim().to_uppercase())
|
||||
.collect();
|
||||
|
||||
// Format 1: SBD in col 0 AND TOAN in col 2 → separate-scores
|
||||
// build-database.js:68: if (cols[0] === "SBD" && cols[2] === "TOAN")
|
||||
if cols.first().map(|s| s.as_str()) == Some("SBD")
|
||||
&& cols.get(2).map(|s| s.as_str()) == Some("TOAN")
|
||||
{
|
||||
return DetectedFormat::SeparateScores;
|
||||
}
|
||||
|
||||
// Format 2: build column index map — check for SOBAODANH|SBD and DIEM_THI
|
||||
// build-database.js:70–86
|
||||
let mut sbd_idx: Option<usize> = None;
|
||||
let mut ho_ten_idx: Option<usize> = None;
|
||||
let mut ngay_sinh_idx: Option<usize> = None;
|
||||
let mut ten_cum_thi_idx: Option<usize> = None;
|
||||
let mut gioi_tinh_idx: Option<usize> = None;
|
||||
let mut diem_thi_idx: Option<usize> = None;
|
||||
|
||||
for (i, c) in cols.iter().enumerate() {
|
||||
match c.as_str() {
|
||||
"SOBAODANH" | "SBD" => sbd_idx = Some(i),
|
||||
"HO_TEN" | "HOTEN" | "HỌ TÊN" => ho_ten_idx = Some(i),
|
||||
"NGAY_SINH" => ngay_sinh_idx = Some(i),
|
||||
"TEN_CUMTHI" => ten_cum_thi_idx = Some(i),
|
||||
"GIOI_TINH" => gioi_tinh_idx = Some(i),
|
||||
"DIEM_THI" => diem_thi_idx = Some(i),
|
||||
_ => {}
|
||||
}
|
||||
}
|
||||
|
||||
// build-database.js:82–84: if (map.sbd !== undefined && map.diem_thi !== undefined)
|
||||
if let (Some(sbd), Some(diem_thi)) = (sbd_idx, diem_thi_idx) {
|
||||
let ho_ten = ho_ten_idx.unwrap_or(1); // fallback: col 1 (present in all known files)
|
||||
return DetectedFormat::Mapped {
|
||||
sbd,
|
||||
ho_ten,
|
||||
ngay_sinh: ngay_sinh_idx,
|
||||
ten_cum_thi: ten_cum_thi_idx,
|
||||
gioi_tinh: gioi_tinh_idx,
|
||||
diem_thi,
|
||||
};
|
||||
}
|
||||
|
||||
// Unrecognised header (or no header) → positional default
|
||||
DetectedFormat::Default
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Row processors
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// Cell accessor helper.
|
||||
fn cell_str(row: &[Data], idx: usize) -> String {
|
||||
row.get(idx)
|
||||
.map(|c| c.to_string().trim().to_owned())
|
||||
.unwrap_or_default()
|
||||
}
|
||||
|
||||
/// Parse a cell that should hold a float score; returns None for blank/non-numeric.
|
||||
/// Mirrors `parseFloat(row[N]) || null` in JS.
|
||||
fn parse_float_cell(row: &[Data], idx: usize) -> Option<f64> {
|
||||
let s = cell_str(row, idx);
|
||||
if s.is_empty() {
|
||||
return None;
|
||||
}
|
||||
s.parse::<f64>().ok().filter(|v| v.is_finite() && *v != 0.0)
|
||||
}
|
||||
|
||||
/// Process a row in the `separate-scores` format.
|
||||
///
|
||||
/// Column layout (build-database.js:90–116 `processSeparateScoresRow`):
|
||||
/// 0=SBD 1=HOTEN 2=TOAN 3=VAN 4=LY 5=HOA 6=SINH 7=SU 8=DIA
|
||||
/// 9=NGOAINGUTN 10=NGOAINGUTL 11=NGOAINGU(total→tieng_anh)
|
||||
///
|
||||
/// tieng_phap / tieng_duc / tieng_nhat / tieng_trung all → None
|
||||
/// ngay_sinh / ten_cum_thi / gioi_tinh all → None (not in this format)
|
||||
pub fn process_separate_scores_row(row: &[Data], patterns: &CompiledPatterns) -> Option<ParsedRow> {
|
||||
let sbd = cell_str(row, 0);
|
||||
let ho_ten = cell_str(row, 1);
|
||||
if sbd.is_empty() || ho_ten.is_empty() {
|
||||
return None;
|
||||
}
|
||||
|
||||
let ho_ten_ascii = to_ascii(&ho_ten);
|
||||
|
||||
// build-database.js:102–114: explicit per-column score mapping
|
||||
let mut scores = std::collections::HashMap::new();
|
||||
macro_rules! add_score {
|
||||
($field:expr, $idx:expr) => {
|
||||
if let Some(v) = parse_float_cell(row, $idx) {
|
||||
scores.insert($field.to_string(), v);
|
||||
}
|
||||
};
|
||||
}
|
||||
add_score!("toan", 2);
|
||||
add_score!("ngu_van", 3);
|
||||
add_score!("vat_ly", 4);
|
||||
add_score!("hoa_hoc", 5);
|
||||
add_score!("sinh_hoc", 6);
|
||||
add_score!("lich_su", 7);
|
||||
add_score!("dia_ly", 8);
|
||||
// col 11 = NGOAINGU total → tieng_anh (build-database.js:110–111)
|
||||
add_score!("tieng_anh", 11);
|
||||
|
||||
// Suppress unused-variable warning; patterns not used in this path (no DIEM_THI string)
|
||||
let _ = patterns;
|
||||
|
||||
Some(ParsedRow {
|
||||
so_bao_danh: sbd,
|
||||
ho_ten,
|
||||
ho_ten_ascii,
|
||||
ngay_sinh: None,
|
||||
ten_cum_thi: None,
|
||||
gioi_tinh: None,
|
||||
scores,
|
||||
})
|
||||
}
|
||||
|
||||
/// Process a row in the `mapped` format (header-derived column indices).
|
||||
///
|
||||
/// Mirrors `processMappedRow` at build-database.js:119–146.
|
||||
/// Gender is normalised: only "Nam" or "Nữ" are kept; everything else → None.
|
||||
/// (build-database.js:132: `(rawGioiTinh === "Nam" || rawGioiTinh === "Nữ") ? rawGioiTinh : null`)
|
||||
pub fn process_mapped_row(
|
||||
row: &[Data],
|
||||
sbd_idx: usize,
|
||||
ho_ten_idx: usize,
|
||||
ngay_sinh_idx: Option<usize>,
|
||||
ten_cum_thi_idx: Option<usize>,
|
||||
gioi_tinh_idx: Option<usize>,
|
||||
diem_thi_idx: usize,
|
||||
patterns: &CompiledPatterns,
|
||||
) -> Option<ParsedRow> {
|
||||
let sbd = cell_str(row, sbd_idx);
|
||||
let ho_ten = cell_str(row, ho_ten_idx);
|
||||
if sbd.is_empty() || ho_ten.is_empty() {
|
||||
return None;
|
||||
}
|
||||
|
||||
// Skip rows where SBD or HO_TEN are themselves header tokens (leaked header rows).
|
||||
// build-database.js:125–126: KNOWN_HEADERS.has(sbdUpper) || KNOWN_HEADERS.has(hoTenUpper)
|
||||
let sbd_upper = sbd.to_uppercase();
|
||||
let ho_ten_upper = ho_ten.to_uppercase();
|
||||
if KNOWN_HEADERS.iter().any(|h| *h == sbd_upper.as_str())
|
||||
|| KNOWN_HEADERS.iter().any(|h| *h == ho_ten_upper.as_str())
|
||||
{
|
||||
return None;
|
||||
}
|
||||
|
||||
let ho_ten_ascii = to_ascii(&ho_ten);
|
||||
|
||||
let ngay_sinh = ngay_sinh_idx
|
||||
.map(|i| cell_str(row, i))
|
||||
.filter(|s| !s.is_empty());
|
||||
|
||||
let ten_cum_thi = ten_cum_thi_idx
|
||||
.map(|i| cell_str(row, i))
|
||||
.filter(|s| !s.is_empty());
|
||||
|
||||
// build-database.js:130–132: normalise gender
|
||||
let gioi_tinh = gioi_tinh_idx
|
||||
.map(|i| cell_str(row, i))
|
||||
.and_then(|s| {
|
||||
if s == "Nam" || s == "Nữ" {
|
||||
Some(s)
|
||||
} else {
|
||||
None
|
||||
}
|
||||
});
|
||||
|
||||
let diem_thi = row
|
||||
.get(diem_thi_idx)
|
||||
.map(|c| c.to_string())
|
||||
.unwrap_or_default();
|
||||
|
||||
let scores = parse_scores(&diem_thi, patterns);
|
||||
|
||||
Some(ParsedRow {
|
||||
so_bao_danh: sbd,
|
||||
ho_ten,
|
||||
ho_ten_ascii,
|
||||
ngay_sinh,
|
||||
ten_cum_thi,
|
||||
gioi_tinh,
|
||||
scores,
|
||||
})
|
||||
}
|
||||
|
||||
/// Process a row using the default positional 6-column layout.
|
||||
///
|
||||
/// Column order: SBD(0) HO_TEN(1) NGAY_SINH(2) TEN_CUMTHI(3) GIOI_TINH(4) DIEM_THI(5)
|
||||
/// Mirrors `processMappedRow(row, DEFAULT_MAP)` at build-database.js:233–236.
|
||||
pub fn process_default_row(row: &[Data], patterns: &CompiledPatterns) -> Option<ParsedRow> {
|
||||
process_mapped_row(
|
||||
row,
|
||||
0, // sbd
|
||||
1, // ho_ten
|
||||
Some(2),
|
||||
Some(3),
|
||||
Some(4),
|
||||
5, // diem_thi
|
||||
patterns,
|
||||
)
|
||||
}
|
||||
|
||||
/// Dispatch a data row through the correct processor for the detected format.
|
||||
///
|
||||
/// Returns `None` when the row is empty/invalid and should be skipped.
|
||||
pub fn process_row_2016(
|
||||
row: &[Data],
|
||||
fmt: &DetectedFormat,
|
||||
patterns: &CompiledPatterns,
|
||||
) -> Option<ParsedRow> {
|
||||
match fmt {
|
||||
DetectedFormat::SeparateScores => process_separate_scores_row(row, patterns),
|
||||
DetectedFormat::Mapped {
|
||||
sbd,
|
||||
ho_ten,
|
||||
ngay_sinh,
|
||||
ten_cum_thi,
|
||||
gioi_tinh,
|
||||
diem_thi,
|
||||
} => process_mapped_row(
|
||||
row,
|
||||
*sbd,
|
||||
*ho_ten,
|
||||
*ngay_sinh,
|
||||
*ten_cum_thi,
|
||||
*gioi_tinh,
|
||||
*diem_thi,
|
||||
patterns,
|
||||
),
|
||||
DetectedFormat::Default => process_default_row(row, patterns),
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Unit tests — 3 detection branches + key processing cases
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use std::collections::HashMap;
|
||||
|
||||
fn s(v: &str) -> Data {
|
||||
Data::String(v.to_string())
|
||||
}
|
||||
|
||||
fn make_patterns() -> CompiledPatterns {
|
||||
let mut map = HashMap::new();
|
||||
map.insert("toan".into(), r"Toán:\s*(\d+(?:\.\d+)?)".into());
|
||||
map.insert("ngu_van".into(), r"Ngữ văn:\s*(\d+(?:\.\d+)?)".into());
|
||||
map.insert("tieng_anh".into(), r"Tiếng Anh:\s*(\d+(?:\.\d+)?)".into());
|
||||
map.insert("tieng_duc".into(), r"Tiếng Đức:\s*(\d+(?:\.\d+)?)".into());
|
||||
map.insert("tieng_nhat".into(), r"Tiếng Nhật:\s*(\d+(?:\.\d+)?)".into());
|
||||
CompiledPatterns::new(&map).unwrap()
|
||||
}
|
||||
|
||||
// --- detect_format: branch 1 — separate-scores ---
|
||||
|
||||
#[test]
|
||||
fn detect_separate_scores() {
|
||||
let header = vec![s("SBD"), s("HOTEN"), s("TOAN"), s("VAN")];
|
||||
match detect_format(&header) {
|
||||
DetectedFormat::SeparateScores => {}
|
||||
other => panic!("expected SeparateScores, got {other:?}"),
|
||||
}
|
||||
}
|
||||
|
||||
// --- detect_format: branch 2 — mapped ---
|
||||
|
||||
#[test]
|
||||
fn detect_mapped_with_named_cols() {
|
||||
let header = vec![
|
||||
s("SOBAODANH"),
|
||||
s("HO_TEN"),
|
||||
s("NGAY_SINH"),
|
||||
s("TEN_CUMTHI"),
|
||||
s("GIOI_TINH"),
|
||||
s("DIEM_THI"),
|
||||
];
|
||||
match detect_format(&header) {
|
||||
DetectedFormat::Mapped {
|
||||
sbd,
|
||||
ho_ten,
|
||||
ngay_sinh,
|
||||
ten_cum_thi,
|
||||
gioi_tinh,
|
||||
diem_thi,
|
||||
} => {
|
||||
assert_eq!(sbd, 0);
|
||||
assert_eq!(ho_ten, 1);
|
||||
assert_eq!(ngay_sinh, Some(2));
|
||||
assert_eq!(ten_cum_thi, Some(3));
|
||||
assert_eq!(gioi_tinh, Some(4));
|
||||
assert_eq!(diem_thi, 5);
|
||||
}
|
||||
other => panic!("expected Mapped, got {other:?}"),
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn detect_mapped_sbd_variant() {
|
||||
// "SBD" (not "SOBAODANH") + DIEM_THI at different positions
|
||||
let header = vec![s("STT"), s("SBD"), s("HOTEN"), s("DIEM_THI")];
|
||||
match detect_format(&header) {
|
||||
DetectedFormat::Mapped { sbd, diem_thi, .. } => {
|
||||
assert_eq!(sbd, 1);
|
||||
assert_eq!(diem_thi, 3);
|
||||
}
|
||||
other => panic!("expected Mapped, got {other:?}"),
|
||||
}
|
||||
}
|
||||
|
||||
// --- detect_format: branch 3 — default ---
|
||||
|
||||
#[test]
|
||||
fn detect_default_when_no_header() {
|
||||
// A data row: positional default used
|
||||
let data_row = vec![
|
||||
s("12345678"),
|
||||
s("Nguyễn Văn A"),
|
||||
s("01/01/2000"),
|
||||
s("TP HCM"),
|
||||
s("Nam"),
|
||||
s("Toán: 8.5"),
|
||||
];
|
||||
// Default is returned when there is no recognised header
|
||||
match detect_format(&data_row) {
|
||||
DetectedFormat::Default => {}
|
||||
other => panic!("expected Default, got {other:?}"),
|
||||
}
|
||||
}
|
||||
|
||||
// --- process_separate_scores_row ---
|
||||
|
||||
#[test]
|
||||
fn separate_scores_basic() {
|
||||
let p = make_patterns();
|
||||
// 12 columns: SBD HOTEN TOAN VAN LY HOA SINH SU DIA NGUTN NGUTL NGUTOTAL
|
||||
let row = vec![
|
||||
s("TP001"),
|
||||
s("Nguyễn Thị Lan"),
|
||||
Data::Float(8.0),
|
||||
Data::Float(7.5),
|
||||
Data::Float(9.0),
|
||||
Data::Float(6.5),
|
||||
Data::Float(5.0),
|
||||
Data::Float(4.5),
|
||||
Data::Float(8.0),
|
||||
Data::Empty,
|
||||
Data::Empty,
|
||||
Data::Float(7.0), // col 11 → tieng_anh
|
||||
];
|
||||
let row = process_separate_scores_row(&row, &p).expect("should parse");
|
||||
assert_eq!(row.so_bao_danh, "TP001");
|
||||
assert_eq!(row.ho_ten, "Nguyễn Thị Lan");
|
||||
assert_eq!(row.ho_ten_ascii, "nguyen thi lan");
|
||||
assert_eq!(row.scores.get("toan"), Some(&8.0));
|
||||
assert_eq!(row.scores.get("tieng_anh"), Some(&7.0));
|
||||
assert!(row.ngay_sinh.is_none());
|
||||
assert!(row.ten_cum_thi.is_none());
|
||||
assert!(row.gioi_tinh.is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn separate_scores_skips_empty_sbd() {
|
||||
let p = make_patterns();
|
||||
let row = vec![s(""), s("Nguyễn Văn A"), Data::Float(5.0)];
|
||||
assert!(process_separate_scores_row(&row, &p).is_none());
|
||||
}
|
||||
|
||||
// --- process_mapped_row ---
|
||||
|
||||
#[test]
|
||||
fn mapped_row_full_fields() {
|
||||
let p = make_patterns();
|
||||
let row = vec![
|
||||
s("HCM001"),
|
||||
s("Trần Thị Bích"),
|
||||
s("15/3/1999"),
|
||||
s("Cụm thi HCM"),
|
||||
s("Nữ"),
|
||||
s("Toán: 9.0 Ngữ văn: 8.5 Tiếng Anh: 7.75"),
|
||||
];
|
||||
let parsed = process_mapped_row(&row, 0, 1, Some(2), Some(3), Some(4), 5, &p)
|
||||
.expect("should parse");
|
||||
assert_eq!(parsed.so_bao_danh, "HCM001");
|
||||
assert_eq!(parsed.ngay_sinh.as_deref(), Some("15/3/1999"));
|
||||
assert_eq!(parsed.ten_cum_thi.as_deref(), Some("Cụm thi HCM"));
|
||||
assert_eq!(parsed.gioi_tinh.as_deref(), Some("Nữ"));
|
||||
assert_eq!(parsed.scores.get("toan"), Some(&9.0));
|
||||
assert_eq!(parsed.scores.get("ngu_van"), Some(&8.5));
|
||||
assert_eq!(parsed.scores.get("tieng_anh"), Some(&7.75));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn mapped_row_gender_normalisation() {
|
||||
let p = make_patterns();
|
||||
// Gender "Unknown" → None
|
||||
let row = vec![s("ABC"), s("Nguyen Van A"), s(""), s(""), s("Unknown"), s("")];
|
||||
let parsed =
|
||||
process_mapped_row(&row, 0, 1, Some(2), Some(3), Some(4), 5, &p).expect("should parse");
|
||||
assert!(parsed.gioi_tinh.is_none());
|
||||
|
||||
// "Nam" passes through
|
||||
let row2 = vec![s("ABC"), s("Nguyen Van A"), s(""), s(""), s("Nam"), s("")];
|
||||
let parsed2 =
|
||||
process_mapped_row(&row2, 0, 1, Some(2), Some(3), Some(4), 5, &p).expect("should parse");
|
||||
assert_eq!(parsed2.gioi_tinh.as_deref(), Some("Nam"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn mapped_row_skips_leaked_header() {
|
||||
let p = make_patterns();
|
||||
// A leaked header row — SBD cell contains "SOBAODANH"
|
||||
let row = vec![s("SOBAODANH"), s("HO_TEN"), s("NGAY_SINH"), s(""), s(""), s("")];
|
||||
assert!(process_mapped_row(&row, 0, 1, Some(2), Some(3), Some(4), 5, &p).is_none());
|
||||
}
|
||||
|
||||
// --- process_default_row ---
|
||||
|
||||
#[test]
|
||||
fn default_row_positional() {
|
||||
let p = make_patterns();
|
||||
let row = vec![
|
||||
s("DN001"),
|
||||
s("Lê Văn Long"),
|
||||
s("10/5/1998"),
|
||||
s("Cụm Đà Nẵng"),
|
||||
s("Nam"),
|
||||
s("Tiếng Đức: 6.25"),
|
||||
];
|
||||
let parsed = process_default_row(&row, &p).expect("should parse");
|
||||
assert_eq!(parsed.so_bao_danh, "DN001");
|
||||
assert_eq!(parsed.ho_ten_ascii, "le van long");
|
||||
assert_eq!(parsed.ngay_sinh.as_deref(), Some("10/5/1998"));
|
||||
assert_eq!(parsed.ten_cum_thi.as_deref(), Some("Cụm Đà Nẵng"));
|
||||
assert_eq!(parsed.gioi_tinh.as_deref(), Some("Nam"));
|
||||
assert_eq!(parsed.scores.get("tieng_duc"), Some(&6.25));
|
||||
}
|
||||
|
||||
// --- is_header_row_2016 ---
|
||||
|
||||
#[test]
|
||||
fn header_row_detection_2016() {
|
||||
assert!(is_header_row_2016(&[s("SOBAODANH"), s("HO_TEN")]));
|
||||
assert!(is_header_row_2016(&[s("SBD"), s("HOTEN"), s("TOAN")]));
|
||||
assert!(!is_header_row_2016(&[s("12345678"), s("Nguyen Van A")]));
|
||||
assert!(!is_header_row_2016(&[s("SBD")])); // too short (< 2 cells)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,10 @@
|
||||
/// Public library interface for integration tests.
|
||||
/// The binary entry point is src/main.rs; this file re-exports the internal
|
||||
/// modules so tests/golden.rs can call them without going through the CLI.
|
||||
pub mod audit;
|
||||
pub mod config;
|
||||
pub mod error;
|
||||
pub mod format_detect_2016;
|
||||
pub mod reader;
|
||||
pub mod transform;
|
||||
pub mod writer;
|
||||
@@ -0,0 +1,379 @@
|
||||
/// xlsxread — Rust CLI replacing the SheetJS xlsx build scripts.
|
||||
///
|
||||
/// Subcommands:
|
||||
/// build — read .xls/.xlsx files → write SQLite DB
|
||||
/// audit — compare distinct SBD count from xlsx vs DB row count
|
||||
///
|
||||
/// Library modules are declared in lib.rs; main.rs only adds the CLI layer.
|
||||
///
|
||||
/// When config contains `format_detection = "thptqg2016"` the build subcommand
|
||||
/// uses per-file header inspection to pick the right column layout, replicating
|
||||
/// the `detectFormat` logic from scripts/build-database.js (lines 63–87).
|
||||
mod cli;
|
||||
|
||||
use std::path::Path;
|
||||
|
||||
use anyhow::{Context, Result};
|
||||
use calamine::Data;
|
||||
use clap::Parser;
|
||||
|
||||
use cli::{Cli, Cmd};
|
||||
use xlsxread::audit;
|
||||
use xlsxread::config::load_config;
|
||||
use xlsxread::format_detect_2016::{
|
||||
detect_format, is_header_row_2016, process_row_2016, DetectedFormat,
|
||||
};
|
||||
use xlsxread::reader::{is_all_blank, process_file};
|
||||
use xlsxread::transform::{validate_row, CompiledPatterns, SkipReason};
|
||||
use xlsxread::writer::{
|
||||
finish_db, insert_row, insert_row_2016, open_db, SCORE_FIELDS, SCORE_FIELDS_2016,
|
||||
};
|
||||
|
||||
fn main() -> Result<()> {
|
||||
let cli = Cli::parse();
|
||||
|
||||
match cli.cmd {
|
||||
Cmd::Build {
|
||||
schema,
|
||||
input,
|
||||
output,
|
||||
} => {
|
||||
run_build(&schema, &input, &output)?;
|
||||
}
|
||||
Cmd::Audit { schema, input, db } => {
|
||||
let cfg = load_config(&schema)
|
||||
.with_context(|| format!("Failed to load config: {}", schema.display()))?;
|
||||
let result = audit::run_audit(&input, &db, &cfg).with_context(|| "Audit failed")?;
|
||||
audit::print_audit_report(&result);
|
||||
if !result.matched {
|
||||
std::process::exit(1);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Build subcommand — dispatches to thptqg2016 or standard path
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
fn run_build(schema_path: &Path, input_dir: &Path, output_path: &Path) -> Result<()> {
|
||||
let cfg = load_config(schema_path)
|
||||
.with_context(|| format!("Failed to load config: {}", schema_path.display()))?;
|
||||
|
||||
if cfg.format_detection.as_deref() == Some("thptqg2016") {
|
||||
run_build_2016(&cfg, input_dir, output_path)
|
||||
} else {
|
||||
run_build_standard(&cfg, input_dir, output_path)
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Standard build path (thptqg2017 and similar fixed-column configs)
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
fn run_build_standard(
|
||||
cfg: &xlsxread::config::DatasetConfig,
|
||||
input_dir: &Path,
|
||||
output_path: &Path,
|
||||
) -> Result<()> {
|
||||
let patterns =
|
||||
CompiledPatterns::new(&cfg.scores).with_context(|| "Failed to compile score regexes")?;
|
||||
|
||||
let mut files: Vec<std::path::PathBuf> = std::fs::read_dir(input_dir)
|
||||
.with_context(|| format!("Cannot read input dir: {}", input_dir.display()))?
|
||||
.filter_map(|e| e.ok())
|
||||
.map(|e| e.path())
|
||||
.filter(|p| {
|
||||
p.is_file()
|
||||
&& p.extension()
|
||||
.and_then(|e| e.to_str())
|
||||
.map(|e| {
|
||||
let lower = e.to_lowercase();
|
||||
lower == "xls" || lower == "xlsx"
|
||||
})
|
||||
.unwrap_or(false)
|
||||
})
|
||||
.collect();
|
||||
files.sort();
|
||||
|
||||
let dataset_label = input_dir
|
||||
.file_name()
|
||||
.and_then(|n| n.to_str())
|
||||
.unwrap_or("data");
|
||||
|
||||
println!(
|
||||
"[build] {dataset_label}/ → {} ({} files)",
|
||||
output_path.display(),
|
||||
files.len()
|
||||
);
|
||||
|
||||
let conn = open_db(output_path, cfg)
|
||||
.with_context(|| format!("Failed to open DB: {}", output_path.display()))?;
|
||||
|
||||
let mut total_source_rows: u64 = 0;
|
||||
let mut total_skipped: u64 = 0;
|
||||
let mut total_errors: u64 = 0;
|
||||
|
||||
let is_old2 = dataset_label.contains("old2");
|
||||
let strip_blank = cfg.reader.strip_blank_rows;
|
||||
|
||||
conn.execute_batch("BEGIN")?;
|
||||
|
||||
for file in &files {
|
||||
let base = file
|
||||
.file_name()
|
||||
.and_then(|n| n.to_str())
|
||||
.unwrap_or("?")
|
||||
.to_owned();
|
||||
let mut file_rows: u64 = 0;
|
||||
let mut file_skipped: u64 = 0;
|
||||
let mut file_errors: u64 = 0;
|
||||
|
||||
let process_result = process_file(file, cfg, |_sheet_idx, raw| {
|
||||
let all_blank = is_all_blank(raw);
|
||||
if strip_blank && all_blank {
|
||||
return;
|
||||
}
|
||||
|
||||
total_source_rows += 1;
|
||||
|
||||
let ho_ten = raw
|
||||
.get(cfg.columns.as_ref().unwrap().ho_ten)
|
||||
.map(|c| c.to_string().trim().to_owned())
|
||||
.unwrap_or_default();
|
||||
let so_bao_danh = raw
|
||||
.get(cfg.columns.as_ref().unwrap().so_bao_danh)
|
||||
.map(|c| c.to_string().trim().to_owned())
|
||||
.unwrap_or_default();
|
||||
|
||||
match validate_row(&ho_ten, &so_bao_danh, &cfg.validation, strip_blank, all_blank) {
|
||||
Err(SkipReason::BlankRow) => {}
|
||||
Err(_) => {
|
||||
file_skipped += 1;
|
||||
return;
|
||||
}
|
||||
Ok(()) => {}
|
||||
}
|
||||
|
||||
let parsed = xlsxread::transform::transform_row(raw, cfg, &patterns);
|
||||
|
||||
match insert_row(&conn, &cfg.insert.sql, &parsed, SCORE_FIELDS) {
|
||||
Ok(()) => file_rows += 1,
|
||||
Err(e) => {
|
||||
file_errors += 1;
|
||||
if total_errors + file_errors <= 5 {
|
||||
eprintln!(" [warn] {base}: {e}");
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
|
||||
match process_result {
|
||||
Ok(_) => {}
|
||||
Err(e) => {
|
||||
eprintln!(" [error] {base}: {e}");
|
||||
file_errors += 1;
|
||||
}
|
||||
}
|
||||
|
||||
total_skipped += file_skipped;
|
||||
total_errors += file_errors;
|
||||
println!(" {base}: {file_rows} rows");
|
||||
}
|
||||
|
||||
conn.execute_batch("COMMIT")?;
|
||||
|
||||
finish_db(
|
||||
&conn,
|
||||
output_path,
|
||||
total_source_rows,
|
||||
total_skipped,
|
||||
total_errors,
|
||||
dataset_label,
|
||||
files.len(),
|
||||
is_old2,
|
||||
)
|
||||
.with_context(|| "Failed to finalise DB")?;
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// thptqg2016 build path — per-file format detection
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// Build the thptqg2016 database.
|
||||
///
|
||||
/// Each file is processed independently: the first row is inspected to determine
|
||||
/// which of the three column layouts applies (separate-scores / mapped / default).
|
||||
/// This mirrors `detectFormat` in scripts/build-database.js lines 63–87, called
|
||||
/// once per file inside the file loop at build-database.js:218–219.
|
||||
fn run_build_2016(
|
||||
cfg: &xlsxread::config::DatasetConfig,
|
||||
input_dir: &Path,
|
||||
output_path: &Path,
|
||||
) -> Result<()> {
|
||||
let patterns =
|
||||
CompiledPatterns::new(&cfg.scores).with_context(|| "Failed to compile score regexes")?;
|
||||
|
||||
let mut files: Vec<std::path::PathBuf> = std::fs::read_dir(input_dir)
|
||||
.with_context(|| format!("Cannot read input dir: {}", input_dir.display()))?
|
||||
.filter_map(|e| e.ok())
|
||||
.map(|e| e.path())
|
||||
.filter(|p| {
|
||||
p.is_file()
|
||||
&& p.extension()
|
||||
.and_then(|e| e.to_str())
|
||||
.map(|e| {
|
||||
let lower = e.to_lowercase();
|
||||
lower == "xls" || lower == "xlsx"
|
||||
})
|
||||
.unwrap_or(false)
|
||||
})
|
||||
.collect();
|
||||
files.sort();
|
||||
|
||||
let dataset_label = input_dir
|
||||
.file_name()
|
||||
.and_then(|n| n.to_str())
|
||||
.unwrap_or("data");
|
||||
|
||||
println!(
|
||||
"[build:2016] {dataset_label}/ → {} ({} files)",
|
||||
output_path.display(),
|
||||
files.len()
|
||||
);
|
||||
|
||||
let conn = open_db(output_path, cfg)
|
||||
.with_context(|| format!("Failed to open DB: {}", output_path.display()))?;
|
||||
|
||||
let mut total_source_rows: u64 = 0;
|
||||
let total_skipped: u64 = 0;
|
||||
let mut total_errors: u64 = 0;
|
||||
|
||||
conn.execute_batch("BEGIN")?;
|
||||
|
||||
for file in &files {
|
||||
let base = file
|
||||
.file_name()
|
||||
.and_then(|n| n.to_str())
|
||||
.unwrap_or("?")
|
||||
.to_owned();
|
||||
|
||||
match process_file_2016(
|
||||
file,
|
||||
cfg,
|
||||
&patterns,
|
||||
&conn,
|
||||
&base,
|
||||
&mut total_source_rows,
|
||||
&mut total_errors,
|
||||
) {
|
||||
Ok(file_rows) => {
|
||||
println!(" {base}: {file_rows} rows");
|
||||
}
|
||||
Err(e) => {
|
||||
eprintln!(" [error] {base}: {e}");
|
||||
total_errors += 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
conn.execute_batch("COMMIT")?;
|
||||
|
||||
finish_db(
|
||||
&conn,
|
||||
output_path,
|
||||
total_source_rows,
|
||||
total_skipped,
|
||||
total_errors,
|
||||
dataset_label,
|
||||
files.len(),
|
||||
false,
|
||||
)
|
||||
.with_context(|| "Failed to finalise DB")?;
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Process one file in the thptqg2016 format-detection path.
|
||||
///
|
||||
/// Reads the file, uses the first row to detect the column layout, then processes
|
||||
/// all subsequent data rows. Returns the count of successfully inserted rows.
|
||||
fn process_file_2016(
|
||||
file: &Path,
|
||||
cfg: &xlsxread::config::DatasetConfig,
|
||||
patterns: &CompiledPatterns,
|
||||
conn: &rusqlite::Connection,
|
||||
base: &str,
|
||||
total_source_rows: &mut u64,
|
||||
total_errors: &mut u64,
|
||||
) -> Result<u64> {
|
||||
use calamine::{open_workbook_auto, Reader, Sheets};
|
||||
|
||||
let path_str = file.display().to_string();
|
||||
let mut workbook: Sheets<_> =
|
||||
open_workbook_auto(file).with_context(|| format!("Cannot open {path_str}"))?;
|
||||
|
||||
let sheet_names: Vec<String> = workbook.sheet_names().to_vec();
|
||||
if sheet_names.is_empty() {
|
||||
return Ok(0);
|
||||
}
|
||||
|
||||
// Sheet selection: thptqg2016 data/ has all-sheets mode to handle
|
||||
// HCM/HN overflow (same reason as thptqg2017 data/).
|
||||
let sheets_to_read: Vec<String> = match cfg.reader.sheet_mode {
|
||||
xlsxread::config::SheetMode::All => sheet_names.clone(),
|
||||
xlsxread::config::SheetMode::First => vec![sheet_names[0].clone()],
|
||||
};
|
||||
|
||||
let mut file_rows: u64 = 0;
|
||||
|
||||
for sheet_name in &sheets_to_read {
|
||||
let range = workbook
|
||||
.worksheet_range(sheet_name)
|
||||
.with_context(|| format!("Cannot read sheet {sheet_name} in {path_str}"))?;
|
||||
|
||||
let rows: Vec<Vec<Data>> = range.rows().map(|r| r.to_vec()).collect();
|
||||
if rows.is_empty() {
|
||||
continue;
|
||||
}
|
||||
|
||||
// Detect format from first row, then determine start index.
|
||||
// Mirrors build-database.js:215–220: isHeaderRow check + detectFormat.
|
||||
let (fmt, start_idx) = if is_header_row_2016(&rows[0]) {
|
||||
(detect_format(&rows[0]), 1)
|
||||
} else {
|
||||
(DetectedFormat::Default, 0)
|
||||
};
|
||||
|
||||
for row in rows.iter().skip(start_idx) {
|
||||
if row.len() < 2 {
|
||||
continue;
|
||||
}
|
||||
|
||||
*total_source_rows += 1;
|
||||
|
||||
match process_row_2016(row, &fmt, patterns) {
|
||||
None => {
|
||||
// Row was empty/invalid — skipped (mirrors JS `if (!record) continue`)
|
||||
}
|
||||
Some(parsed) => {
|
||||
match insert_row_2016(conn, &cfg.insert.sql, &parsed, SCORE_FIELDS_2016) {
|
||||
Ok(()) => file_rows += 1,
|
||||
Err(e) => {
|
||||
*total_errors += 1;
|
||||
if *total_errors <= 5 {
|
||||
eprintln!(" [warn] {base}: {e}");
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Ok(file_rows)
|
||||
}
|
||||
@@ -0,0 +1,197 @@
|
||||
/// Spreadsheet reader: wraps calamine to iterate rows across sheets.
|
||||
///
|
||||
/// Sheet selection mirrors the JS scripts:
|
||||
/// - sheet_mode = "all" → iterate every sheet (handles HCM/HN 65k overflow in data/)
|
||||
/// - sheet_mode = "first" → sheet 0 only (data-old/)
|
||||
///
|
||||
/// Header detection mirrors build-lib.js isHeaderRow:
|
||||
/// row[0].toUpperCase() in {"HO_TEN", "HỌ TÊN", "STT"}
|
||||
use std::path::Path;
|
||||
|
||||
use calamine::{open_workbook_auto, Data, Reader, Sheets};
|
||||
|
||||
use crate::config::{DatasetConfig, HeaderCfg, SheetMode};
|
||||
use crate::error::BuildError;
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Public row representation from calamine
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
pub type RawRow = Vec<Data>;
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Header detection — mirrors build-lib.js isHeaderRow
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// Returns true when the first cell (uppercased) matches one of the configured
|
||||
/// header tokens. Used to skip the header row on the first row of each sheet.
|
||||
pub fn is_header_row(row: &[Data], header_cfg: &HeaderCfg) -> bool {
|
||||
if row.len() < 3 {
|
||||
return false;
|
||||
}
|
||||
let first = row[0].to_string().trim().to_uppercase();
|
||||
header_cfg.tokens.iter().any(|t| t.to_uppercase() == first)
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// All-blank row check (data-old2: strip_blank_rows)
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
pub fn is_all_blank(row: &[Data]) -> bool {
|
||||
row.iter()
|
||||
.all(|c| matches!(c, Data::Empty) || c.to_string().trim().is_empty())
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// File processor — yields all data rows from the file
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// Process one spreadsheet file, calling `on_row` for each data row.
|
||||
///
|
||||
/// `on_row` receives `(sheet_index, row_index_in_sheet, raw_row)` where
|
||||
/// `row_index_in_sheet` is 0-based AFTER the header has been consumed.
|
||||
/// Returns `(sheets_seen, total_rows_yielded)`.
|
||||
pub fn process_file<F>(
|
||||
path: &Path,
|
||||
cfg: &DatasetConfig,
|
||||
mut on_row: F,
|
||||
) -> Result<(usize, usize), BuildError>
|
||||
where
|
||||
F: FnMut(usize, &RawRow),
|
||||
{
|
||||
let path_str = path.display().to_string();
|
||||
|
||||
// calamine::open_workbook_auto dispatches on file extension
|
||||
let mut workbook: Sheets<_> = open_workbook_auto(path).map_err(|e| BuildError::Calamine {
|
||||
path: path_str.clone(),
|
||||
source: e,
|
||||
})?;
|
||||
|
||||
let sheet_names: Vec<String> = workbook.sheet_names().to_vec();
|
||||
if sheet_names.is_empty() {
|
||||
return Err(BuildError::NoSheets(path_str.clone()));
|
||||
}
|
||||
|
||||
// Sheet selection per config
|
||||
let sheets_to_read: Vec<String> = match cfg.reader.sheet_mode {
|
||||
SheetMode::All => sheet_names.clone(),
|
||||
SheetMode::First => vec![sheet_names[0].clone()],
|
||||
};
|
||||
|
||||
let mut total_rows = 0usize;
|
||||
|
||||
for (sheet_idx, sheet_name) in sheets_to_read.iter().enumerate() {
|
||||
let range = workbook
|
||||
.worksheet_range(sheet_name)
|
||||
.map_err(|e| BuildError::Calamine {
|
||||
path: path_str.clone(),
|
||||
source: e,
|
||||
})?;
|
||||
|
||||
let mut first_row = true;
|
||||
|
||||
for raw in range.rows() {
|
||||
let row: RawRow = raw.to_vec();
|
||||
|
||||
// Skip header row on first row of each sheet (matches JS: `if (i === 0 && isHeaderRow(...))`)
|
||||
if first_row {
|
||||
first_row = false;
|
||||
if is_header_row(&row, &cfg.header) {
|
||||
continue;
|
||||
}
|
||||
}
|
||||
|
||||
on_row(sheet_idx, &row);
|
||||
total_rows += 1;
|
||||
}
|
||||
}
|
||||
|
||||
Ok((sheets_to_read.len(), total_rows))
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Unit tests
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use crate::config::HeaderCfg;
|
||||
|
||||
fn hdr(tokens: &[&str]) -> HeaderCfg {
|
||||
HeaderCfg {
|
||||
tokens: tokens.iter().map(|s| s.to_string()).collect(),
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn header_detects_ho_ten() {
|
||||
let row = vec![
|
||||
Data::String("HO_TEN".into()),
|
||||
Data::String("NGAY_SINH".into()),
|
||||
Data::String("SBD".into()),
|
||||
];
|
||||
let cfg = hdr(&["HO_TEN", "HỌ TÊN", "STT"]);
|
||||
assert!(is_header_row(&row, &cfg));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn header_detects_stt() {
|
||||
let row = vec![
|
||||
Data::String("STT".into()),
|
||||
Data::String("B".into()),
|
||||
Data::String("C".into()),
|
||||
];
|
||||
let cfg = hdr(&["HO_TEN", "HỌ TÊN", "STT"]);
|
||||
assert!(is_header_row(&row, &cfg));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn header_detects_ho_ten_unicode() {
|
||||
let row = vec![
|
||||
Data::String("HỌ TÊN".into()),
|
||||
Data::String("B".into()),
|
||||
Data::String("C".into()),
|
||||
];
|
||||
let cfg = hdr(&["HO_TEN", "HỌ TÊN", "STT"]);
|
||||
assert!(is_header_row(&row, &cfg));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn header_rejects_data_row() {
|
||||
let row = vec![
|
||||
Data::String("Nguyen Van A".into()),
|
||||
Data::String("01/01/2000".into()),
|
||||
Data::String("12345678".into()),
|
||||
];
|
||||
let cfg = hdr(&["HO_TEN", "HỌ TÊN", "STT"]);
|
||||
assert!(!is_header_row(&row, &cfg));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn header_rejects_short_row() {
|
||||
let row = vec![Data::String("HO_TEN".into()), Data::Empty];
|
||||
let cfg = hdr(&["HO_TEN", "HỌ TÊN", "STT"]);
|
||||
assert!(!is_header_row(&row, &cfg));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn header_case_insensitive() {
|
||||
let row = vec![
|
||||
Data::String("ho_ten".into()),
|
||||
Data::String("B".into()),
|
||||
Data::String("C".into()),
|
||||
];
|
||||
let cfg = hdr(&["HO_TEN", "HỌ TÊN", "STT"]);
|
||||
assert!(is_header_row(&row, &cfg));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn blank_row_detection() {
|
||||
let row = vec![Data::Empty, Data::Empty, Data::String("".into())];
|
||||
assert!(is_all_blank(&row));
|
||||
|
||||
let row2 = vec![Data::String("Nguyen".into()), Data::Empty, Data::Empty];
|
||||
assert!(!is_all_blank(&row2));
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,410 @@
|
||||
/// Row transformation: ascii normalisation, score regex parsing, validation.
|
||||
///
|
||||
/// `to_ascii` replicates build-lib.js `toAscii` exactly:
|
||||
/// str.normalize("NFD").replace(/[̀-ͯ]/g,"").replace(/đ/gi,"d").toLowerCase()
|
||||
use std::collections::HashMap;
|
||||
|
||||
use regex::Regex;
|
||||
use unicode_normalization::UnicodeNormalization;
|
||||
|
||||
use crate::config::{DatasetConfig, ValidationCfg};
|
||||
use crate::error::BuildError;
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Compiled score patterns (built once at startup from config)
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
pub struct CompiledPatterns {
|
||||
/// Ordered list so INSERT column order is deterministic
|
||||
pub patterns: Vec<(String, Regex)>,
|
||||
}
|
||||
|
||||
impl CompiledPatterns {
|
||||
pub fn new(scores: &HashMap<String, String>) -> Result<Self, BuildError> {
|
||||
let mut patterns = Vec::with_capacity(scores.len());
|
||||
for (field, src) in scores {
|
||||
let re = Regex::new(src).map_err(|e| BuildError::Regex {
|
||||
pattern: src.clone(),
|
||||
source: e,
|
||||
})?;
|
||||
patterns.push((field.clone(), re));
|
||||
}
|
||||
// Sort for deterministic order across HashMap iteration
|
||||
patterns.sort_by(|a, b| a.0.cmp(&b.0));
|
||||
Ok(Self { patterns })
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// to_ascii — must be byte-for-byte equivalent to build-lib.js toAscii
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// Normalise a Vietnamese name to an ASCII slug.
|
||||
///
|
||||
/// Algorithm mirrors the JavaScript `toAscii` in build-lib.js:
|
||||
/// 1. NFD decompose (splits base + combining diacritics)
|
||||
/// 2. Drop all Unicode combining marks (U+0300–U+036F)
|
||||
/// 3. Replace đ/Đ with d (NFD does not decompose đ)
|
||||
/// 4. Lowercase
|
||||
pub fn to_ascii(s: &str) -> String {
|
||||
// Step 1 + 2: NFD then filter out combining marks (Unicode category M)
|
||||
let decomposed: String = s
|
||||
.nfd()
|
||||
.filter(|c| !('\u{0300}'..='\u{036f}').contains(c))
|
||||
.collect();
|
||||
|
||||
// Step 3: đ/Đ are not decomposed by NFD — replace explicitly
|
||||
let replaced = decomposed.replace(['đ', 'Đ'], "d");
|
||||
|
||||
// Step 4: lowercase
|
||||
replaced.to_lowercase()
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Parsed row ready for DB insert
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
pub struct ParsedRow {
|
||||
pub so_bao_danh: String,
|
||||
pub ho_ten: String,
|
||||
pub ho_ten_ascii: String,
|
||||
pub ngay_sinh: Option<String>,
|
||||
/// thptqg2016 only: examination cluster name (TEN_CUMTHI column)
|
||||
pub ten_cum_thi: Option<String>,
|
||||
/// thptqg2016 only: gender (GIOI_TINH column), normalised to "Nam"/"Nữ" or None
|
||||
pub gioi_tinh: Option<String>,
|
||||
/// Subject field → float value; absent subjects not in map → NULL
|
||||
pub scores: HashMap<String, f64>,
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Row validation — mirrors the per-script skip logic
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// Returns `None` when the row should be skipped entirely (before sourceRows counter).
|
||||
/// Returns `Some(reason)` when the row should be counted as sourceRows but skipped.
|
||||
#[derive(Debug, PartialEq, Eq)]
|
||||
pub enum SkipReason {
|
||||
/// Row is fully blank (data-old2 only, before sourceRows counter)
|
||||
BlankRow,
|
||||
/// soBaoDanh or hoTen empty/missing
|
||||
EmptyField,
|
||||
/// soBaoDanh contains non-digit characters (data-old / data-old2 guard)
|
||||
NonNumericSbd,
|
||||
}
|
||||
|
||||
/// Validates a raw cell slice against the dataset's `ValidationCfg`.
|
||||
/// Returns `Ok(())` on pass, `Err(SkipReason)` on fail.
|
||||
pub fn validate_row(
|
||||
ho_ten: &str,
|
||||
so_bao_danh: &str,
|
||||
cfg: &ValidationCfg,
|
||||
strip_blank_rows: bool,
|
||||
all_blank: bool,
|
||||
) -> Result<(), SkipReason> {
|
||||
// data-old2: skip fully blank rows BEFORE counting sourceRows
|
||||
if strip_blank_rows && all_blank {
|
||||
return Err(SkipReason::BlankRow);
|
||||
}
|
||||
|
||||
if cfg.require_nonempty_sbd && so_bao_danh.is_empty() {
|
||||
return Err(SkipReason::EmptyField);
|
||||
}
|
||||
if cfg.require_nonempty_name && ho_ten.is_empty() {
|
||||
return Err(SkipReason::EmptyField);
|
||||
}
|
||||
if cfg.require_numeric_sbd && !so_bao_danh.chars().all(|c| c.is_ascii_digit()) {
|
||||
return Err(SkipReason::NonNumericSbd);
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Score parsing — mirrors build-lib.js parseScores
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// Parse a DIEM_THI cell string and extract matching subject scores.
|
||||
pub fn parse_scores(diem_thi: &str, patterns: &CompiledPatterns) -> HashMap<String, f64> {
|
||||
let mut out = HashMap::new();
|
||||
for (field, re) in &patterns.patterns {
|
||||
if let Some(caps) = re.captures(diem_thi) {
|
||||
if let Some(m) = caps.get(1) {
|
||||
if let Ok(v) = m.as_str().parse::<f64>() {
|
||||
if v.is_finite() {
|
||||
out.insert(field.clone(), v);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
out
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Full row transform (thptqg2017 fixed-column path)
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// Extract and transform one spreadsheet row into a `ParsedRow` using fixed column indices.
|
||||
/// `raw` is the full cell slice; column indices come from `cfg.columns`.
|
||||
/// Used for thptqg2017 configs that have a static [columns] table.
|
||||
pub fn transform_row(
|
||||
raw: &[calamine::Data],
|
||||
cfg: &DatasetConfig,
|
||||
patterns: &CompiledPatterns,
|
||||
) -> ParsedRow {
|
||||
let cols = cfg
|
||||
.columns
|
||||
.as_ref()
|
||||
.expect("transform_row requires [columns] section in config");
|
||||
|
||||
let get = |idx: usize| -> String {
|
||||
raw.get(idx)
|
||||
.map(|cell| cell.to_string().trim().to_owned())
|
||||
.unwrap_or_default()
|
||||
};
|
||||
|
||||
let ho_ten = get(cols.ho_ten);
|
||||
let ngay_sinh = get(cols.ngay_sinh);
|
||||
let so_bao_danh = get(cols.so_bao_danh);
|
||||
let diem_thi = raw
|
||||
.get(cols.diem_thi)
|
||||
.map(|c| c.to_string())
|
||||
.unwrap_or_default();
|
||||
|
||||
let ho_ten_ascii = to_ascii(&ho_ten);
|
||||
let scores = parse_scores(&diem_thi, patterns);
|
||||
let ngay_sinh_opt = if ngay_sinh.is_empty() {
|
||||
None
|
||||
} else {
|
||||
Some(ngay_sinh)
|
||||
};
|
||||
|
||||
ParsedRow {
|
||||
so_bao_danh,
|
||||
ho_ten,
|
||||
ho_ten_ascii,
|
||||
ngay_sinh: ngay_sinh_opt,
|
||||
ten_cum_thi: None,
|
||||
gioi_tinh: None,
|
||||
scores,
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Unit tests — 20 cases for to_ascii (real Vietnamese names)
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
// Helper: assert to_ascii(input) == expected
|
||||
fn check(input: &str, expected: &str) {
|
||||
assert_eq!(
|
||||
to_ascii(input),
|
||||
expected,
|
||||
"to_ascii({input:?}) expected {expected:?}"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ascii_plain_latin() {
|
||||
check("Nguyen Van A", "nguyen van a");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ascii_nguyen_thi_hoa() {
|
||||
check("Nguyễn Thị Hoa", "nguyen thi hoa");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ascii_tran_van_duc() {
|
||||
// đ/Đ replacement
|
||||
check("Trần Văn Đức", "tran van duc");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ascii_le_thi_my_duyen() {
|
||||
check("Lê Thị Mỹ Duyên", "le thi my duyen");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ascii_pham_thi_lan() {
|
||||
check("Phạm Thị Lan", "pham thi lan");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ascii_bui_thi_thu() {
|
||||
check("Bùi Thị Thu", "bui thi thu");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ascii_hoang_van_truong() {
|
||||
check("Hoàng Văn Trường", "hoang van truong");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ascii_do_thi_ngan() {
|
||||
// Đ uppercase at start
|
||||
check("Đỗ Thị Ngân", "do thi ngan");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ascii_nguyen_van_khanh() {
|
||||
check("Nguyễn Văn Khánh", "nguyen van khanh");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ascii_trinh_thi_bich_ngoc() {
|
||||
check("Trịnh Thị Bích Ngọc", "trinh thi bich ngoc");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ascii_vu_thi_dieu() {
|
||||
// ề = e + combining grave + combining circumflex (after NFD)
|
||||
check("Vũ Thị Diệu", "vu thi dieu");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ascii_nguyen_thi_tuong_vi() {
|
||||
check("Nguyễn Thị Tường Vi", "nguyen thi tuong vi");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ascii_lowercase_d_stroke() {
|
||||
// Lowercase đ → d
|
||||
check("đặng thị hằng", "dang thi hang");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ascii_uppercase_d_stroke() {
|
||||
check("ĐẶNG THỊ HẰNG", "dang thi hang");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ascii_mixed_case() {
|
||||
check("NGUYỄN VĂN AN", "nguyen van an");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ascii_tran_thi_kim_anh() {
|
||||
check("Trần Thị Kim Anh", "tran thi kim anh");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ascii_nguyen_thi_phuong_thao() {
|
||||
check("Nguyễn Thị Phương Thảo", "nguyen thi phuong thao");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ascii_le_van_long() {
|
||||
check("Lê Văn Long", "le van long");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ascii_vo_thi_xuan_mai() {
|
||||
check("Võ Thị Xuân Mai", "vo thi xuan mai");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ascii_empty_string() {
|
||||
check("", "");
|
||||
}
|
||||
|
||||
// --- Score parsing tests ---
|
||||
|
||||
fn make_patterns() -> CompiledPatterns {
|
||||
let mut map = HashMap::new();
|
||||
map.insert("toan".into(), r"Toán:\s*(\d+(?:\.\d+)?)".into());
|
||||
map.insert("ngu_van".into(), r"Ngữ văn:\s*(\d+(?:\.\d+)?)".into());
|
||||
map.insert("vat_ly".into(), r"Vật lí:\s*(\d+(?:\.\d+)?)".into());
|
||||
CompiledPatterns::new(&map).unwrap()
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn parse_scores_single() {
|
||||
let p = make_patterns();
|
||||
let s = "Toán: 8.5";
|
||||
let scores = parse_scores(s, &p);
|
||||
assert_eq!(scores.get("toan"), Some(&8.5));
|
||||
assert!(scores.get("ngu_van").is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn parse_scores_multiple() {
|
||||
let p = make_patterns();
|
||||
let s = "Toán: 7.25 Ngữ văn: 6.0 Vật lí: 9";
|
||||
let scores = parse_scores(s, &p);
|
||||
assert_eq!(scores.get("toan"), Some(&7.25));
|
||||
assert_eq!(scores.get("ngu_van"), Some(&6.0));
|
||||
assert_eq!(scores.get("vat_ly"), Some(&9.0));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn parse_scores_empty_cell() {
|
||||
let p = make_patterns();
|
||||
let scores = parse_scores("", &p);
|
||||
assert!(scores.is_empty());
|
||||
}
|
||||
|
||||
// --- Validation tests ---
|
||||
|
||||
fn default_validation() -> ValidationCfg {
|
||||
ValidationCfg {
|
||||
require_numeric_sbd: false,
|
||||
require_nonempty_name: true,
|
||||
require_nonempty_sbd: true,
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn validate_ok() {
|
||||
let v = default_validation();
|
||||
assert!(validate_row("Nguyen Van A", "12345678", &v, false, false).is_ok());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn validate_empty_sbd() {
|
||||
let v = default_validation();
|
||||
assert_eq!(
|
||||
validate_row("Nguyen Van A", "", &v, false, false),
|
||||
Err(SkipReason::EmptyField)
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn validate_empty_name() {
|
||||
let v = default_validation();
|
||||
assert_eq!(
|
||||
validate_row("", "12345678", &v, false, false),
|
||||
Err(SkipReason::EmptyField)
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn validate_non_numeric_sbd_rejected() {
|
||||
let mut v = default_validation();
|
||||
v.require_numeric_sbd = true;
|
||||
assert_eq!(
|
||||
validate_row("Nguyen Van A", "12AB5678", &v, false, false),
|
||||
Err(SkipReason::NonNumericSbd)
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn validate_numeric_sbd_accepted() {
|
||||
let mut v = default_validation();
|
||||
v.require_numeric_sbd = true;
|
||||
assert!(validate_row("Nguyen Van A", "12345678", &v, false, false).is_ok());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn validate_blank_row_skipped() {
|
||||
let v = default_validation();
|
||||
// strip_blank_rows=true AND all_blank=true → BlankRow
|
||||
assert_eq!(
|
||||
validate_row("", "", &v, true, true),
|
||||
Err(SkipReason::BlankRow)
|
||||
);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,201 @@
|
||||
/// SQLite writer: DDL setup, batched INSERT OR REPLACE, VACUUM, stats output.
|
||||
///
|
||||
/// Mirrors build-lib.js createDb + the transaction loop in each build-database*.js.
|
||||
/// Stats output lines match the JS stdout exactly so existing CI log-greps still work.
|
||||
///
|
||||
/// thptqg2016 differences vs thptqg2017:
|
||||
/// - INSERT includes ten_cum_thi and gioi_tinh after ngay_sinh
|
||||
/// - Score columns are toan/ngu_van/vat_ly/hoa_hoc/sinh_hoc/lich_su/dia_ly/
|
||||
/// tieng_anh/tieng_phap/tieng_duc/tieng_nhat/tieng_trung
|
||||
/// (no khtn, khxh, tieng_nga)
|
||||
use std::fs;
|
||||
use std::path::Path;
|
||||
|
||||
use rusqlite::{params_from_iter, Connection, ToSql};
|
||||
|
||||
use crate::config::DatasetConfig;
|
||||
use crate::error::BuildError;
|
||||
use crate::transform::ParsedRow;
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// DB initialisation — mirrors build-lib.js createDb (delete + recreate)
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// Open (or recreate) the output SQLite database, execute the DDL from config,
|
||||
/// and return the open connection ready for inserts.
|
||||
pub fn open_db(db_path: &Path, cfg: &DatasetConfig) -> Result<Connection, BuildError> {
|
||||
// Mirror Node behaviour: delete existing file before creating (build-lib.js:54)
|
||||
if db_path.exists() {
|
||||
fs::remove_file(db_path).map_err(|e| BuildError::Io {
|
||||
path: db_path.display().to_string(),
|
||||
source: e,
|
||||
})?;
|
||||
}
|
||||
|
||||
// Ensure parent directory exists
|
||||
if let Some(parent) = db_path.parent() {
|
||||
if !parent.as_os_str().is_empty() {
|
||||
fs::create_dir_all(parent).map_err(|e| BuildError::Io {
|
||||
path: parent.display().to_string(),
|
||||
source: e,
|
||||
})?;
|
||||
}
|
||||
}
|
||||
|
||||
let conn = Connection::open(db_path)?;
|
||||
conn.execute_batch(&cfg.schema.ddl)?;
|
||||
Ok(conn)
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Score field lists — one per dataset schema
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// thptqg2017 score columns (14 fields including khtn/khxh/tieng_nga).
|
||||
pub const SCORE_FIELDS_2017: &[&str] = &[
|
||||
"toan",
|
||||
"ngu_van",
|
||||
"vat_ly",
|
||||
"hoa_hoc",
|
||||
"sinh_hoc",
|
||||
"khtn",
|
||||
"lich_su",
|
||||
"dia_ly",
|
||||
"gdcd",
|
||||
"khxh",
|
||||
"tieng_anh",
|
||||
"tieng_phap",
|
||||
"tieng_nga",
|
||||
"tieng_trung",
|
||||
];
|
||||
|
||||
/// thptqg2016 score columns (12 fields: tieng_duc/tieng_nhat present; no khtn/khxh/tieng_nga).
|
||||
pub const SCORE_FIELDS_2016: &[&str] = &[
|
||||
"toan",
|
||||
"ngu_van",
|
||||
"vat_ly",
|
||||
"hoa_hoc",
|
||||
"sinh_hoc",
|
||||
"lich_su",
|
||||
"dia_ly",
|
||||
"tieng_anh",
|
||||
"tieng_phap",
|
||||
"tieng_duc",
|
||||
"tieng_nhat",
|
||||
"tieng_trung",
|
||||
];
|
||||
|
||||
/// Alias kept so thptqg2017 callers that import SCORE_FIELDS continue to compile.
|
||||
pub const SCORE_FIELDS: &[&str] = SCORE_FIELDS_2017;
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Insert a single parsed row (thptqg2017 schema — no ten_cum_thi / gioi_tinh)
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// Bind all fields from `row` into the prepared statement and execute it.
|
||||
/// Positional params: so_bao_danh, ho_ten, ho_ten_ascii, ngay_sinh, <scores...>
|
||||
pub fn insert_row(
|
||||
conn: &Connection,
|
||||
sql: &str,
|
||||
row: &ParsedRow,
|
||||
score_fields: &[&str],
|
||||
) -> Result<(), BuildError> {
|
||||
let mut params: Vec<Box<dyn ToSql>> = Vec::with_capacity(4 + score_fields.len());
|
||||
params.push(Box::new(row.so_bao_danh.clone()));
|
||||
params.push(Box::new(row.ho_ten.clone()));
|
||||
params.push(Box::new(row.ho_ten_ascii.clone()));
|
||||
params.push(Box::new(row.ngay_sinh.clone()));
|
||||
|
||||
for field in score_fields {
|
||||
let val: Option<f64> = row.scores.get(*field).copied();
|
||||
params.push(Box::new(val));
|
||||
}
|
||||
|
||||
conn.execute(sql, params_from_iter(params.iter().map(|p| p.as_ref())))?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Insert a single parsed row (thptqg2016 schema — includes ten_cum_thi + gioi_tinh)
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// Bind all fields from `row` into the prepared statement for the thptqg2016 schema.
|
||||
/// Positional params: so_bao_danh, ho_ten, ho_ten_ascii, ngay_sinh, ten_cum_thi,
|
||||
/// gioi_tinh, <scores...>
|
||||
pub fn insert_row_2016(
|
||||
conn: &Connection,
|
||||
sql: &str,
|
||||
row: &ParsedRow,
|
||||
score_fields: &[&str],
|
||||
) -> Result<(), BuildError> {
|
||||
let mut params: Vec<Box<dyn ToSql>> = Vec::with_capacity(6 + score_fields.len());
|
||||
params.push(Box::new(row.so_bao_danh.clone()));
|
||||
params.push(Box::new(row.ho_ten.clone()));
|
||||
params.push(Box::new(row.ho_ten_ascii.clone()));
|
||||
params.push(Box::new(row.ngay_sinh.clone()));
|
||||
params.push(Box::new(row.ten_cum_thi.clone()));
|
||||
params.push(Box::new(row.gioi_tinh.clone()));
|
||||
|
||||
for field in score_fields {
|
||||
let val: Option<f64> = row.scores.get(*field).copied();
|
||||
params.push(Box::new(val));
|
||||
}
|
||||
|
||||
conn.execute(sql, params_from_iter(params.iter().map(|p| p.as_ref())))?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Post-build: VACUUM + stats output
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// Run VACUUM and print statistics lines that mirror the Node scripts' stdout.
|
||||
/// The exact prefix tokens ("Source data rows", "DB rows", "Size:") are preserved
|
||||
/// so any log-grep in the deploy pipeline keeps working.
|
||||
#[allow(clippy::too_many_arguments)]
|
||||
pub fn finish_db(
|
||||
conn: &Connection,
|
||||
db_path: &Path,
|
||||
source_rows: u64,
|
||||
skipped: u64,
|
||||
errors: u64,
|
||||
dataset_label: &str,
|
||||
_file_count: usize,
|
||||
is_old2: bool,
|
||||
) -> Result<(), BuildError> {
|
||||
conn.execute_batch("VACUUM")?;
|
||||
|
||||
let db_count: i64 = conn.query_row("SELECT COUNT(*) FROM student", [], |row| row.get(0))?;
|
||||
|
||||
let insertable = source_rows - skipped;
|
||||
|
||||
println!();
|
||||
if is_old2 {
|
||||
println!("Source non-blank data rows: {source_rows}");
|
||||
println!(" skipped (empty/non-numeric SBD): {skipped}");
|
||||
} else {
|
||||
println!("Source data rows (post-header): {source_rows}");
|
||||
if dataset_label.contains("old") {
|
||||
println!(" skipped (empty/non-numeric SBD): {skipped}");
|
||||
} else {
|
||||
println!(" skipped (empty/invalid): {skipped}");
|
||||
}
|
||||
}
|
||||
println!(" insertable: {insertable}");
|
||||
println!(" insert errors: {errors}");
|
||||
println!("DB rows (distinct SBD): {db_count}");
|
||||
|
||||
if !dataset_label.contains("old") && errors == 0 {
|
||||
let gap = insertable as i64 - db_count;
|
||||
if gap == 0 {
|
||||
println!("Audit: OK — every source row made it in.");
|
||||
} else {
|
||||
println!("Audit: {gap} row(s) collapsed (duplicate SBDs overwriting).");
|
||||
}
|
||||
}
|
||||
|
||||
let sz = fs::metadata(db_path).map(|m| m.len()).unwrap_or(0);
|
||||
println!("Size: {:.1} MB", sz as f64 / 1024.0 / 1024.0);
|
||||
|
||||
Ok(())
|
||||
}
|
||||
+18
@@ -0,0 +1,18 @@
|
||||
# Test Fixtures
|
||||
|
||||
Anonymised `.xlsx` files for integration testing. All student PII has been replaced:
|
||||
|
||||
- `ho_ten` replaced with `Nguyen Van Test NNN` / `Tran Thi Test NNN` patterns
|
||||
- `so_bao_danh` replaced with sequential synthetic numbers (e.g. `10000001`)
|
||||
- `ngay_sinh` replaced with fixed synthetic dates
|
||||
- Scores are realistic random values in the 0–10 range
|
||||
|
||||
Files:
|
||||
|
||||
- `province-100.xlsx` — 100-row single-sheet file (simulates a normal province)
|
||||
- `hcm-overflow.xlsx` — 2-sheet file (200 rows Sheet1 + 200 rows Sheet2, simulating HCM overflow)
|
||||
- `province-numeric-sbd.xlsx` — 100 rows with strictly numeric SBDs (for data-old variant)
|
||||
|
||||
These files are generated by `tests/golden.rs` `generate_fixtures()` if they do not already exist
|
||||
on disk. The generator is pure Rust (uses the `zip` crate already pulled in via calamine).
|
||||
No external Python or Node tooling required for unit/integration tests.
|
||||
Binary file not shown.
Binary file not shown.
Binary file not shown.
@@ -0,0 +1,691 @@
|
||||
/// Stage 5 golden tests — integration tests using anonymised fixture files.
|
||||
///
|
||||
/// Fixture files are generated in-process via raw OOXML + zip if they do not
|
||||
/// already exist on disk. No external Python or Node tooling required for the
|
||||
/// Rust-side tests. The Node golden comparison is marked #[ignore] when pnpm
|
||||
/// is not in PATH.
|
||||
use std::io::Write as IoWrite;
|
||||
use std::path::{Path, PathBuf};
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Minimal OOXML xlsx generator
|
||||
//
|
||||
// Produces a valid .xlsx that calamine can read. Only uses the `zip` crate
|
||||
// which is already pulled in as a transitive dependency of calamine.
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// One row of cell data for a fixture sheet.
|
||||
struct XlsxRow {
|
||||
values: Vec<String>,
|
||||
}
|
||||
|
||||
/// Write a minimal .xlsx to `path` with the given sheets.
|
||||
/// `sheets`: Vec<(sheet_name, rows)> where rows[0] is the header.
|
||||
fn write_xlsx(path: &Path, sheets: &[(String, Vec<XlsxRow>)]) {
|
||||
use zip::{write::SimpleFileOptions, ZipWriter};
|
||||
|
||||
let file = std::fs::File::create(path).expect("create fixture xlsx");
|
||||
let mut zip = ZipWriter::new(file);
|
||||
let opts = SimpleFileOptions::default();
|
||||
|
||||
// [Content_Types].xml
|
||||
let mut content_types = String::from(
|
||||
r#"<?xml version="1.0" encoding="UTF-8" standalone="yes"?>
|
||||
<Types xmlns="http://schemas.openxmlformats.org/package/2006/content-types">
|
||||
<Default Extension="rels" ContentType="application/vnd.openxmlformats-package.relationships+xml"/>
|
||||
<Default Extension="xml" ContentType="application/xml"/>
|
||||
<Override PartName="/xl/workbook.xml" ContentType="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet.main+xml"/>
|
||||
"#,
|
||||
);
|
||||
for (i, _) in sheets.iter().enumerate() {
|
||||
content_types.push_str(&format!(
|
||||
r#" <Override PartName="/xl/worksheets/sheet{}.xml" ContentType="application/vnd.openxmlformats-officedocument.spreadsheetml.worksheet+xml"/>
|
||||
"#,
|
||||
i + 1
|
||||
));
|
||||
}
|
||||
content_types.push_str("</Types>");
|
||||
zip.start_file("[Content_Types].xml", opts).unwrap();
|
||||
zip.write_all(content_types.as_bytes()).unwrap();
|
||||
|
||||
// _rels/.rels
|
||||
zip.start_file("_rels/.rels", opts).unwrap();
|
||||
zip.write_all(
|
||||
br#"<?xml version="1.0" encoding="UTF-8" standalone="yes"?>
|
||||
<Relationships xmlns="http://schemas.openxmlformats.org/package/2006/relationships">
|
||||
<Relationship Id="rId1" Type="http://schemas.openxmlformats.org/officeDocument/2006/relationships/officeDocument" Target="xl/workbook.xml"/>
|
||||
</Relationships>"#,
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
// xl/_rels/workbook.xml.rels
|
||||
let mut wb_rels = String::from(
|
||||
r#"<?xml version="1.0" encoding="UTF-8" standalone="yes"?>
|
||||
<Relationships xmlns="http://schemas.openxmlformats.org/package/2006/relationships">
|
||||
"#,
|
||||
);
|
||||
for (i, _) in sheets.iter().enumerate() {
|
||||
wb_rels.push_str(&format!(
|
||||
r#" <Relationship Id="rId{}" Type="http://schemas.openxmlformats.org/officeDocument/2006/relationships/worksheet" Target="worksheets/sheet{}.xml"/>
|
||||
"#,
|
||||
i + 1,
|
||||
i + 1
|
||||
));
|
||||
}
|
||||
wb_rels.push_str("</Relationships>");
|
||||
zip.start_file("xl/_rels/workbook.xml.rels", opts).unwrap();
|
||||
zip.write_all(wb_rels.as_bytes()).unwrap();
|
||||
|
||||
// xl/workbook.xml
|
||||
let mut wb = String::from(
|
||||
r#"<?xml version="1.0" encoding="UTF-8" standalone="yes"?>
|
||||
<workbook xmlns="http://schemas.openxmlformats.org/spreadsheetml/2006/main"
|
||||
xmlns:r="http://schemas.openxmlformats.org/officeDocument/2006/relationships">
|
||||
<sheets>
|
||||
"#,
|
||||
);
|
||||
for (i, (name, _)) in sheets.iter().enumerate() {
|
||||
let escaped = xml_escape(name);
|
||||
wb.push_str(&format!(
|
||||
r#" <sheet name="{}" sheetId="{}" r:id="rId{}"/>
|
||||
"#,
|
||||
escaped,
|
||||
i + 1,
|
||||
i + 1
|
||||
));
|
||||
}
|
||||
wb.push_str(" </sheets>\n</workbook>");
|
||||
zip.start_file("xl/workbook.xml", opts).unwrap();
|
||||
zip.write_all(wb.as_bytes()).unwrap();
|
||||
|
||||
// xl/worksheets/sheetN.xml
|
||||
for (i, (_, rows)) in sheets.iter().enumerate() {
|
||||
let mut ws = String::from(
|
||||
r#"<?xml version="1.0" encoding="UTF-8" standalone="yes"?>
|
||||
<worksheet xmlns="http://schemas.openxmlformats.org/spreadsheetml/2006/main">
|
||||
<sheetData>
|
||||
"#,
|
||||
);
|
||||
for (row_idx, row) in rows.iter().enumerate() {
|
||||
ws.push_str(&format!(
|
||||
r#" <row r="{}">
|
||||
"#,
|
||||
row_idx + 1
|
||||
));
|
||||
for (col_idx, val) in row.values.iter().enumerate() {
|
||||
let col_letter = col_letter(col_idx);
|
||||
let cell_ref = format!("{}{}", col_letter, row_idx + 1);
|
||||
let escaped = xml_escape(val);
|
||||
ws.push_str(&format!(
|
||||
r#" <c r="{}" t="inlineStr"><is><t>{}</t></is></c>
|
||||
"#,
|
||||
cell_ref, escaped
|
||||
));
|
||||
}
|
||||
ws.push_str(" </row>\n");
|
||||
}
|
||||
ws.push_str(" </sheetData>\n</worksheet>");
|
||||
zip.start_file(&format!("xl/worksheets/sheet{}.xml", i + 1), opts)
|
||||
.unwrap();
|
||||
zip.write_all(ws.as_bytes()).unwrap();
|
||||
}
|
||||
|
||||
zip.finish().unwrap();
|
||||
}
|
||||
|
||||
fn col_letter(idx: usize) -> &'static str {
|
||||
const LETTERS: &[&str] = &[
|
||||
"A", "B", "C", "D", "E", "F", "G", "H", "I", "J", "K", "L", "M", "N", "O", "P", "Q", "R",
|
||||
"S", "T", "U", "V", "W", "X", "Y", "Z",
|
||||
];
|
||||
LETTERS[idx % 26]
|
||||
}
|
||||
|
||||
fn xml_escape(s: &str) -> String {
|
||||
s.replace('&', "&")
|
||||
.replace('<', "<")
|
||||
.replace('>', ">")
|
||||
.replace('"', """)
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Fixture data builders
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
fn header_row() -> XlsxRow {
|
||||
XlsxRow {
|
||||
values: vec![
|
||||
"HO_TEN".into(),
|
||||
"NGAY_SINH".into(),
|
||||
"SO_BAO_DANH".into(),
|
||||
"DIEM_THI".into(),
|
||||
],
|
||||
}
|
||||
}
|
||||
|
||||
fn data_row(idx: usize, scores: &str) -> XlsxRow {
|
||||
// Anonymised: name uses sequential pattern, SBD is purely synthetic
|
||||
let name = if idx % 2 == 0 {
|
||||
format!("Nguyen Van Test {:03}", idx)
|
||||
} else {
|
||||
format!("Tran Thi Test {:03}", idx)
|
||||
};
|
||||
XlsxRow {
|
||||
values: vec![
|
||||
name,
|
||||
format!("15/0{}/{}", (idx % 9) + 1, 1999 + (idx % 5)),
|
||||
format!("1000{:04}", idx),
|
||||
scores.to_owned(),
|
||||
],
|
||||
}
|
||||
}
|
||||
|
||||
fn sample_scores(idx: usize) -> String {
|
||||
// Realistic scores in 0–10 range, varies by idx
|
||||
let toan = 4.0 + (idx % 60) as f64 / 10.0;
|
||||
let van = 3.5 + (idx % 65) as f64 / 10.0;
|
||||
format!("Toán: {toan:.1} Ngữ văn: {van:.1} Tiếng Anh: 7.5")
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Fixture file paths
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
fn fixtures_dir() -> PathBuf {
|
||||
// tests/fixtures/ relative to the crate root
|
||||
let mut p = PathBuf::from(env!("CARGO_MANIFEST_DIR"));
|
||||
p.push("tests");
|
||||
p.push("fixtures");
|
||||
p
|
||||
}
|
||||
|
||||
fn province_fixture_path() -> PathBuf {
|
||||
fixtures_dir().join("province-100.xlsx")
|
||||
}
|
||||
|
||||
fn hcm_overflow_fixture_path() -> PathBuf {
|
||||
fixtures_dir().join("hcm-overflow.xlsx")
|
||||
}
|
||||
|
||||
fn numeric_sbd_fixture_path() -> PathBuf {
|
||||
fixtures_dir().join("province-numeric-sbd.xlsx")
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Fixture generation — called once per test run if files missing
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
fn ensure_fixtures() {
|
||||
let dir = fixtures_dir();
|
||||
std::fs::create_dir_all(&dir).expect("create fixtures dir");
|
||||
|
||||
// province-100.xlsx — 100 data rows, single sheet, with header
|
||||
if !province_fixture_path().exists() {
|
||||
let mut rows = vec![header_row()];
|
||||
for i in 0..100 {
|
||||
rows.push(data_row(i, &sample_scores(i)));
|
||||
}
|
||||
write_xlsx(&province_fixture_path(), &[("Sheet1".to_owned(), rows)]);
|
||||
}
|
||||
|
||||
// hcm-overflow.xlsx — 2 sheets × 200 rows each (no header on sheet 2)
|
||||
if !hcm_overflow_fixture_path().exists() {
|
||||
let mut sheet1 = vec![header_row()];
|
||||
for i in 0..200 {
|
||||
sheet1.push(data_row(i, &sample_scores(i)));
|
||||
}
|
||||
// Sheet2: continuation rows, no header row (as in real HCM overflow)
|
||||
let mut sheet2 = Vec::new();
|
||||
for i in 200..400 {
|
||||
sheet2.push(data_row(i, &sample_scores(i)));
|
||||
}
|
||||
write_xlsx(
|
||||
&hcm_overflow_fixture_path(),
|
||||
&[("Sheet1".to_owned(), sheet1), ("Sheet2".to_owned(), sheet2)],
|
||||
);
|
||||
}
|
||||
|
||||
// province-numeric-sbd.xlsx — strictly numeric SBDs for data-old config
|
||||
if !numeric_sbd_fixture_path().exists() {
|
||||
let mut rows = vec![header_row()];
|
||||
for i in 0..100 {
|
||||
rows.push(XlsxRow {
|
||||
values: vec![
|
||||
format!("Nguyen Van Test {:03}", i),
|
||||
"01/01/2000".to_owned(),
|
||||
format!("{:08}", 20000000 + i), // pure digits
|
||||
sample_scores(i),
|
||||
],
|
||||
});
|
||||
}
|
||||
write_xlsx(&numeric_sbd_fixture_path(), &[("Sheet1".to_owned(), rows)]);
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Config helpers
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
fn make_data_config() -> xlsxread::config::DatasetConfig {
|
||||
let cfg_path = PathBuf::from(env!("CARGO_MANIFEST_DIR"))
|
||||
.join("configs")
|
||||
.join("thptqg2017-data.toml");
|
||||
xlsxread::config::load_config(&cfg_path).expect("load data config")
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Integration tests — pure Rust, no Node dependency
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
#[test]
|
||||
fn province_100_builds_100_rows() {
|
||||
ensure_fixtures();
|
||||
let dir = tempdir();
|
||||
let db_path = dir.join("test.db");
|
||||
let fixture_dir = dir.join("input");
|
||||
std::fs::create_dir_all(&fixture_dir).unwrap();
|
||||
std::fs::copy(province_fixture_path(), fixture_dir.join("province.xlsx")).unwrap();
|
||||
|
||||
run_build_cmd(&fixture_dir, &db_path, "thptqg2017-data.toml");
|
||||
|
||||
let count = query_count(&db_path);
|
||||
assert_eq!(count, 100, "expected 100 rows from province-100 fixture");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn hcm_overflow_builds_400_rows() {
|
||||
ensure_fixtures();
|
||||
let dir = tempdir();
|
||||
let db_path = dir.join("test.db");
|
||||
let fixture_dir = dir.join("input");
|
||||
std::fs::create_dir_all(&fixture_dir).unwrap();
|
||||
std::fs::copy(hcm_overflow_fixture_path(), fixture_dir.join("hcm.xlsx")).unwrap();
|
||||
|
||||
run_build_cmd(&fixture_dir, &db_path, "thptqg2017-data.toml");
|
||||
|
||||
let count = query_count(&db_path);
|
||||
assert_eq!(
|
||||
count, 400,
|
||||
"expected 400 rows (200 × 2 sheets) from hcm-overflow fixture"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn data_old_first_sheet_only_100_rows() {
|
||||
ensure_fixtures();
|
||||
let dir = tempdir();
|
||||
let db_path = dir.join("test.db");
|
||||
let fixture_dir = dir.join("input");
|
||||
std::fs::create_dir_all(&fixture_dir).unwrap();
|
||||
// Use the overflow file but with data-old config (first sheet only → 200 rows)
|
||||
std::fs::copy(hcm_overflow_fixture_path(), fixture_dir.join("hcm.xlsx")).unwrap();
|
||||
|
||||
run_build_cmd(&fixture_dir, &db_path, "thptqg2017-data-old.toml");
|
||||
|
||||
// data-old: sheet_mode=first → only 200 rows from sheet1; but SBDs "1000NNNN" are
|
||||
// all digits so all pass the numeric guard
|
||||
let count = query_count(&db_path);
|
||||
assert_eq!(
|
||||
count, 200,
|
||||
"data-old config should read only first sheet (200 rows)"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn numeric_sbd_guard_rejects_non_numeric() {
|
||||
ensure_fixtures();
|
||||
let dir = tempdir();
|
||||
let db_path = dir.join("test.db");
|
||||
let fixture_dir = dir.join("input");
|
||||
std::fs::create_dir_all(&fixture_dir).unwrap();
|
||||
|
||||
// Write a fixture with one non-numeric SBD mixed in
|
||||
let mut rows = vec![header_row()];
|
||||
for i in 0..10 {
|
||||
rows.push(XlsxRow {
|
||||
values: vec![
|
||||
format!("Test {:03}", i),
|
||||
"01/01/2000".to_owned(),
|
||||
if i == 5 {
|
||||
"ABC123".to_owned()
|
||||
} else {
|
||||
format!("{:08}", 20000000 + i)
|
||||
},
|
||||
sample_scores(i),
|
||||
],
|
||||
});
|
||||
}
|
||||
let mixed_path = fixture_dir.join("mixed.xlsx");
|
||||
write_xlsx(&mixed_path, &[("Sheet1".to_owned(), rows)]);
|
||||
|
||||
run_build_cmd(&fixture_dir, &db_path, "thptqg2017-data-old.toml");
|
||||
|
||||
// Row i=5 has non-numeric SBD → rejected by data-old config
|
||||
let count = query_count(&db_path);
|
||||
assert_eq!(
|
||||
count, 9,
|
||||
"non-numeric SBD row should be skipped by data-old config"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn scores_parsed_correctly_into_db() {
|
||||
ensure_fixtures();
|
||||
let dir = tempdir();
|
||||
let db_path = dir.join("test.db");
|
||||
let fixture_dir = dir.join("input");
|
||||
std::fs::create_dir_all(&fixture_dir).unwrap();
|
||||
|
||||
let rows = vec![
|
||||
header_row(),
|
||||
XlsxRow {
|
||||
values: vec![
|
||||
"Nguyen Van Test 001".to_owned(),
|
||||
"01/01/2000".to_owned(),
|
||||
"10000001".to_owned(),
|
||||
"Toán: 8.5 Ngữ văn: 7.0 Tiếng Anh: 9.25".to_owned(),
|
||||
],
|
||||
},
|
||||
];
|
||||
write_xlsx(
|
||||
&fixture_dir.join("one.xlsx"),
|
||||
&[("Sheet1".to_owned(), rows)],
|
||||
);
|
||||
|
||||
run_build_cmd(&fixture_dir, &db_path, "thptqg2017-data.toml");
|
||||
|
||||
let conn = rusqlite::Connection::open(&db_path).unwrap();
|
||||
let (toan, van, anh): (f64, f64, f64) = conn
|
||||
.query_row(
|
||||
"SELECT toan, ngu_van, tieng_anh FROM student WHERE so_bao_danh = '10000001'",
|
||||
[],
|
||||
|r| Ok((r.get(0)?, r.get(1)?, r.get(2)?)),
|
||||
)
|
||||
.expect("row not found");
|
||||
assert!((toan - 8.5).abs() < 1e-9);
|
||||
assert!((van - 7.0).abs() < 1e-9);
|
||||
assert!((anh - 9.25).abs() < 1e-9);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn to_ascii_stored_correctly() {
|
||||
ensure_fixtures();
|
||||
let dir = tempdir();
|
||||
let db_path = dir.join("test.db");
|
||||
let fixture_dir = dir.join("input");
|
||||
std::fs::create_dir_all(&fixture_dir).unwrap();
|
||||
|
||||
let rows = vec![
|
||||
header_row(),
|
||||
XlsxRow {
|
||||
values: vec![
|
||||
"Nguyễn Văn Đức".to_owned(),
|
||||
"".to_owned(),
|
||||
"20000001".to_owned(),
|
||||
"".to_owned(),
|
||||
],
|
||||
},
|
||||
];
|
||||
write_xlsx(
|
||||
&fixture_dir.join("one.xlsx"),
|
||||
&[("Sheet1".to_owned(), rows)],
|
||||
);
|
||||
|
||||
run_build_cmd(&fixture_dir, &db_path, "thptqg2017-data.toml");
|
||||
|
||||
let conn = rusqlite::Connection::open(&db_path).unwrap();
|
||||
let ascii: String = conn
|
||||
.query_row(
|
||||
"SELECT ho_ten_ascii FROM student WHERE so_bao_danh = '20000001'",
|
||||
[],
|
||||
|r| r.get(0),
|
||||
)
|
||||
.expect("row not found");
|
||||
assert_eq!(ascii, "nguyen van duc");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn audit_subcommand_matches_after_build() {
|
||||
ensure_fixtures();
|
||||
let dir = tempdir();
|
||||
let db_path = dir.join("test.db");
|
||||
let fixture_dir = dir.join("input");
|
||||
std::fs::create_dir_all(&fixture_dir).unwrap();
|
||||
std::fs::copy(province_fixture_path(), fixture_dir.join("province.xlsx")).unwrap();
|
||||
|
||||
run_build_cmd(&fixture_dir, &db_path, "thptqg2017-data.toml");
|
||||
|
||||
// audit should match (100 distinct SBDs in xlsx == 100 rows in DB)
|
||||
let cfg = make_data_config();
|
||||
let result = xlsxread::audit::run_audit(&fixture_dir, &db_path, &cfg).expect("audit failed");
|
||||
assert!(result.matched, "audit should match after build");
|
||||
assert_eq!(result.distinct_sbds, 100);
|
||||
assert_eq!(result.db_count, 100);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn audit_subcommand_mismatch_detected() {
|
||||
ensure_fixtures();
|
||||
let dir = tempdir();
|
||||
let db_path = dir.join("test.db");
|
||||
let fixture_dir = dir.join("input");
|
||||
std::fs::create_dir_all(&fixture_dir).unwrap();
|
||||
|
||||
// Write 10 rows to xlsx but build DB from only 5 rows
|
||||
let mut all_rows = vec![header_row()];
|
||||
for i in 0..10 {
|
||||
all_rows.push(data_row(i, &sample_scores(i)));
|
||||
}
|
||||
write_xlsx(
|
||||
&fixture_dir.join("all.xlsx"),
|
||||
&[("Sheet1".to_owned(), all_rows)],
|
||||
);
|
||||
|
||||
// Build DB with only first 5 rows in a different file
|
||||
let build_dir = dir.join("build_input");
|
||||
std::fs::create_dir_all(&build_dir).unwrap();
|
||||
let mut five_rows = vec![header_row()];
|
||||
for i in 0..5 {
|
||||
five_rows.push(data_row(i, &sample_scores(i)));
|
||||
}
|
||||
write_xlsx(
|
||||
&build_dir.join("five.xlsx"),
|
||||
&[("Sheet1".to_owned(), five_rows)],
|
||||
);
|
||||
|
||||
run_build_cmd(&build_dir, &db_path, "thptqg2017-data.toml");
|
||||
|
||||
// audit against fixture_dir (10 xlsx rows) but DB has 5 rows → mismatch
|
||||
let cfg = make_data_config();
|
||||
let result = xlsxread::audit::run_audit(&fixture_dir, &db_path, &cfg).expect("audit failed");
|
||||
assert!(!result.matched, "audit should not match (10 xlsx vs 5 db)");
|
||||
assert_eq!(result.distinct_sbds, 10);
|
||||
assert_eq!(result.db_count, 5);
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Golden test: compare Rust DB vs Node DB on identical fixture
|
||||
// Marked #[ignore] when pnpm / node is not in PATH — CI installs them first.
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
#[test]
|
||||
#[ignore]
|
||||
fn golden_rust_matches_node_db() {
|
||||
// This test requires: pnpm, node, and the thptqg2017 package to be installed
|
||||
// Run with: cargo test -- --ignored golden_rust_matches_node_db
|
||||
let which_pnpm = std::process::Command::new("which")
|
||||
.arg("pnpm")
|
||||
.output()
|
||||
.map(|o| o.status.success())
|
||||
.unwrap_or(false);
|
||||
if !which_pnpm {
|
||||
eprintln!("pnpm not in PATH — skipping golden test");
|
||||
return;
|
||||
}
|
||||
|
||||
let dir = tempdir();
|
||||
let fixture_dir = dir.join("input");
|
||||
std::fs::create_dir_all(&fixture_dir).unwrap();
|
||||
std::fs::copy(province_fixture_path(), fixture_dir.join("province.xlsx")).unwrap();
|
||||
|
||||
// Build with Rust
|
||||
let rust_db = dir.join("rust.db");
|
||||
run_build_cmd(&fixture_dir, &rust_db, "thptqg2017-data.toml");
|
||||
|
||||
// Build with Node (run build-database.js with DATA_DIR / DB_PATH overrides)
|
||||
// Node script reads env-vars via a thin wrapper — see scripts/build-database.js
|
||||
// For now: diff via SELECT * ORDER BY so_bao_danh
|
||||
let node_db = dir.join("node.db");
|
||||
let status = std::process::Command::new("pnpm")
|
||||
.args(["exec", "node", "scripts/build-database.js"])
|
||||
.env("OVERRIDE_SRC_DIR", fixture_dir.to_str().unwrap())
|
||||
.env("OVERRIDE_DB_PATH", node_db.to_str().unwrap())
|
||||
.current_dir(
|
||||
PathBuf::from(env!("CARGO_MANIFEST_DIR"))
|
||||
.parent()
|
||||
.unwrap()
|
||||
.parent()
|
||||
.unwrap(),
|
||||
)
|
||||
.status()
|
||||
.expect("failed to run node build script");
|
||||
|
||||
if !status.success() {
|
||||
panic!("Node build script failed with: {status}");
|
||||
}
|
||||
|
||||
// Row-by-row comparison
|
||||
diff_dbs(&rust_db, &node_db);
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Helpers
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
fn tempdir() -> PathBuf {
|
||||
let base = std::env::temp_dir().join(format!(
|
||||
"xlsxread-test-{}",
|
||||
std::time::SystemTime::now()
|
||||
.duration_since(std::time::UNIX_EPOCH)
|
||||
.unwrap()
|
||||
.subsec_nanos()
|
||||
));
|
||||
std::fs::create_dir_all(&base).unwrap();
|
||||
base
|
||||
}
|
||||
|
||||
fn run_build_cmd(input_dir: &Path, db_path: &Path, config_name: &str) {
|
||||
let cfg_path = PathBuf::from(env!("CARGO_MANIFEST_DIR"))
|
||||
.join("configs")
|
||||
.join(config_name);
|
||||
|
||||
let cfg = xlsxread::config::load_config(&cfg_path)
|
||||
.unwrap_or_else(|e| panic!("load config {config_name}: {e}"));
|
||||
let patterns =
|
||||
xlsxread::transform::CompiledPatterns::new(&cfg.scores).expect("compile patterns");
|
||||
|
||||
// Collect files
|
||||
let mut files: Vec<PathBuf> = std::fs::read_dir(input_dir)
|
||||
.unwrap()
|
||||
.filter_map(|e| e.ok())
|
||||
.map(|e| e.path())
|
||||
.filter(|p| {
|
||||
p.is_file()
|
||||
&& p.extension()
|
||||
.and_then(|e| e.to_str())
|
||||
.map(|e| {
|
||||
let l = e.to_lowercase();
|
||||
l == "xls" || l == "xlsx"
|
||||
})
|
||||
.unwrap_or(false)
|
||||
})
|
||||
.collect();
|
||||
files.sort();
|
||||
|
||||
let conn = xlsxread::writer::open_db(db_path, &cfg).expect("open db");
|
||||
conn.execute_batch("BEGIN").unwrap();
|
||||
|
||||
for file in &files {
|
||||
xlsxread::reader::process_file(file, &cfg, |_, raw| {
|
||||
let all_blank = xlsxread::reader::is_all_blank(raw);
|
||||
if cfg.reader.strip_blank_rows && all_blank {
|
||||
return;
|
||||
}
|
||||
let cols = cfg.columns.as_ref().expect("golden test requires [columns]");
|
||||
let ho_ten = raw
|
||||
.get(cols.ho_ten)
|
||||
.map(|c| c.to_string().trim().to_owned())
|
||||
.unwrap_or_default();
|
||||
let so_bao_danh = raw
|
||||
.get(cols.so_bao_danh)
|
||||
.map(|c| c.to_string().trim().to_owned())
|
||||
.unwrap_or_default();
|
||||
if xlsxread::transform::validate_row(
|
||||
&ho_ten,
|
||||
&so_bao_danh,
|
||||
&cfg.validation,
|
||||
cfg.reader.strip_blank_rows,
|
||||
all_blank,
|
||||
)
|
||||
.is_err()
|
||||
{
|
||||
return;
|
||||
}
|
||||
let row = xlsxread::transform::transform_row(raw, &cfg, &patterns);
|
||||
let _ = xlsxread::writer::insert_row(
|
||||
&conn,
|
||||
&cfg.insert.sql,
|
||||
&row,
|
||||
xlsxread::writer::SCORE_FIELDS,
|
||||
);
|
||||
})
|
||||
.expect("process file");
|
||||
}
|
||||
|
||||
conn.execute_batch("COMMIT").unwrap();
|
||||
conn.execute_batch("VACUUM").unwrap();
|
||||
}
|
||||
|
||||
fn query_count(db_path: &Path) -> i64 {
|
||||
let conn = rusqlite::Connection::open(db_path).expect("open db for count");
|
||||
conn.query_row("SELECT COUNT(*) FROM student", [], |r| r.get(0))
|
||||
.expect("count query")
|
||||
}
|
||||
|
||||
fn diff_dbs(a: &Path, b: &Path) {
|
||||
let conn_a = rusqlite::Connection::open(a).unwrap();
|
||||
|
||||
// Attach b as "other"
|
||||
conn_a
|
||||
.execute_batch(&format!("ATTACH DATABASE '{}' AS other", b.display()))
|
||||
.unwrap();
|
||||
|
||||
// Rows in a not in b
|
||||
let missing_in_b: i64 = conn_a
|
||||
.query_row(
|
||||
"SELECT COUNT(*) FROM main.student s
|
||||
WHERE NOT EXISTS (SELECT 1 FROM other.student o WHERE o.so_bao_danh = s.so_bao_danh)",
|
||||
[],
|
||||
|r| r.get(0),
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
// Rows in b not in a
|
||||
let missing_in_a: i64 = conn_a
|
||||
.query_row(
|
||||
"SELECT COUNT(*) FROM other.student o
|
||||
WHERE NOT EXISTS (SELECT 1 FROM main.student s WHERE s.so_bao_danh = o.so_bao_danh)",
|
||||
[],
|
||||
|r| r.get(0),
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
assert_eq!(
|
||||
missing_in_b, 0,
|
||||
"{missing_in_b} rows in Rust DB missing from Node DB"
|
||||
);
|
||||
assert_eq!(
|
||||
missing_in_a, 0,
|
||||
"{missing_in_a} rows in Node DB missing from Rust DB"
|
||||
);
|
||||
}
|
||||
Reference in new issue
Block a user