Files
thptqg/2017/tools/xlsxread/src/config.rs
T
tiennm99 99465fda59 feat: replace xlsx (SheetJS) build pipeline with Rust xlsxread CLI (#1)
* feat(xlsxread): stage 0 scaffold with pinned deps and clap CLI skeleton

* feat(xlsxread): stage 1 reader — calamine sheet enumeration and header skip

* feat(xlsxread): stage 2 transform — to_ascii, score regex, validation with 38 unit tests

* feat(xlsxread): stage 3 writer — SQLite DDL, INSERT OR REPLACE, VACUUM, stats output

* feat(xlsxread): stage 4 audit — distinct SBD scan vs DB count, mirrors audit-row-counts.js output

* feat(xlsxread): stage 5 golden tests — in-process xlsx fixtures, 8 integration tests pass

* chore(xlsxread): commit Cargo.lock for reproducible Rust builds

* feat(build): wire root build:db scripts to xlsxread CLI

Replace node scripts/build-database*.js invocations with the Rust
xlsxread binary. Each build:db* script now calls `pnpm build:rust`
(cargo build --release) before invoking the xlsxread build subcommand
with the matching per-dataset config.

Drop xlsx and better-sqlite3 from devDependencies — no Node script
consumes them anymore. sql.js (runtime DB reader in the SPA) is
unaffected and remains in dependencies.

* ci: build xlsxread before running database build jobs

Add dtolnay/rust-toolchain@stable and Swatinem/rust-cache@v2
(workspaces: tools/xlsxread) for warm incremental Rust builds.

Replace the single `pnpm build:db:all` step with explicit xlsxread
invocations so CI doesn't call pnpm build:rust redundantly three times.
The binary is built once, then each of the three datasets is processed
in sequence.

* chore: remove deprecated xlsx-based build scripts

Delete scripts/build-database.js, build-database-old.js,
build-database-old2.js, build-lib.js, and audit-row-counts.js.
Functionality replaced by the xlsxread Rust CLI configured via
tools/xlsxread/configs/*.toml. History preserved in git; one-click
revert available via the chore/migration-backup-260519 branch.

* docs: update README build instructions for xlsxread pipeline

Replace Node.js + xlsx references with Rust + xlsxread workflow.
Update requirements (Node 24+, pnpm, Rust stable), quickstart, scripts
table, and project layout tree to reflect the current state after the
xlsx-based build scripts were removed.

* chore(deps): drop xlsx and better-sqlite3 from package.json and lockfile

Remove xlsx (SheetJS, vulnerable: GHSA-4r6h-8v6p-xvw6, GHSA-5pgg-2g8v-p4x9)
and better-sqlite3 from devDependencies. Both were only used by the now-deleted
Node build scripts. The Rust xlsxread CLI vendors SQLite via rusqlite-bundled;
no Node-side SQLite dependency is needed. `pnpm audit` returns clean.
2026-05-19 16:33:16 +07:00

147 lines
4.1 KiB
Rust

use std::collections::HashMap;
use std::fs;
use std::path::Path;
use serde::Deserialize;
use crate::error::BuildError;
// ---------------------------------------------------------------------------
// Top-level dataset configuration loaded from a .toml file
// ---------------------------------------------------------------------------
#[derive(Debug, Deserialize, Clone)]
pub struct DatasetConfig {
pub reader: ReaderCfg,
pub columns: ColumnMap,
pub validation: ValidationCfg,
pub header: HeaderCfg,
pub schema: SchemaCfg,
/// field name → regex source string (one entry per scoreable subject)
pub scores: HashMap<String, String>,
pub insert: InsertCfg,
}
#[derive(Debug, Deserialize, Clone)]
pub struct ReaderCfg {
/// "all" → iterate every sheet (handles HCM/HN overflow); "first" → sheet 0 only
pub sheet_mode: SheetMode,
/// If true, skip rows where every cell is empty/null before counting (data-old2 quirk)
pub strip_blank_rows: bool,
}
#[derive(Debug, Deserialize, Clone, PartialEq, Eq)]
#[serde(rename_all = "lowercase")]
pub enum SheetMode {
All,
First,
}
/// Zero-indexed column positions in the source spreadsheet row.
#[derive(Debug, Deserialize, Clone)]
pub struct ColumnMap {
pub ho_ten: usize,
pub ngay_sinh: usize,
pub so_bao_danh: usize,
pub diem_thi: usize,
}
#[derive(Debug, Deserialize, Clone)]
pub struct ValidationCfg {
/// build-database-old.js / -old2.js require soBaoDanh to match ^\d+$
pub require_numeric_sbd: bool,
pub require_nonempty_name: bool,
pub require_nonempty_sbd: bool,
}
#[derive(Debug, Deserialize, Clone)]
pub struct HeaderCfg {
/// Tokens to match against row[0].to_uppercase() to detect a header row
pub tokens: Vec<String>,
}
#[derive(Debug, Deserialize, Clone)]
pub struct SchemaCfg {
/// DDL executed verbatim before inserts (CREATE TABLE + CREATE INDEX)
pub ddl: String,
}
#[derive(Debug, Deserialize, Clone)]
pub struct InsertCfg {
/// Parameterised INSERT OR REPLACE SQL using :named_param style
pub sql: String,
}
// ---------------------------------------------------------------------------
// Loader
// ---------------------------------------------------------------------------
pub fn load_config(path: &Path) -> Result<DatasetConfig, BuildError> {
let text = fs::read_to_string(path).map_err(|e| BuildError::Io {
path: path.display().to_string(),
source: e,
})?;
let cfg: DatasetConfig = toml::from_str(&text)?;
Ok(cfg)
}
// ---------------------------------------------------------------------------
// Unit tests
// ---------------------------------------------------------------------------
#[cfg(test)]
mod tests {
use super::*;
const SAMPLE_TOML: &str = r#"
[reader]
sheet_mode = "all"
strip_blank_rows = false
[columns]
ho_ten = 0
ngay_sinh = 1
so_bao_danh = 2
diem_thi = 3
[validation]
require_numeric_sbd = false
require_nonempty_name = true
require_nonempty_sbd = true
[header]
tokens = ["HO_TEN", "HỌ TÊN", "STT"]
[schema]
ddl = "CREATE TABLE student (so_bao_danh TEXT PRIMARY KEY);"
[scores]
toan = 'Toán:\s*(\d+(?:\.\d+)?)'
ngu_van = 'Ngữ văn:\s*(\d+(?:\.\d+)?)'
[insert]
sql = "INSERT OR REPLACE INTO student (so_bao_danh) VALUES (:so_bao_danh)"
"#;
#[test]
fn config_round_trip() {
let cfg: DatasetConfig = toml::from_str(SAMPLE_TOML).expect("parse failed");
assert_eq!(cfg.reader.sheet_mode, SheetMode::All);
assert!(!cfg.reader.strip_blank_rows);
assert_eq!(cfg.columns.ho_ten, 0);
assert_eq!(cfg.columns.diem_thi, 3);
assert!(!cfg.validation.require_numeric_sbd);
assert!(cfg.validation.require_nonempty_name);
assert_eq!(cfg.header.tokens.len(), 3);
assert!(cfg.scores.contains_key("toan"));
assert!(cfg.scores.contains_key("ngu_van"));
}
#[test]
fn config_first_sheet_mode() {
let toml_str = SAMPLE_TOML.replace(r#"sheet_mode = "all""#, r#"sheet_mode = "first""#);
let cfg: DatasetConfig = toml::from_str(&toml_str).expect("parse failed");
assert_eq!(cfg.reader.sheet_mode, SheetMode::First);
}
}