From 76f7d849805b13f894dfc24dcf8480c103ab5345 Mon Sep 17 00:00:00 2001 From: tiennm99 Date: Tue, 19 May 2026 15:10:03 +0700 Subject: [PATCH] =?UTF-8?q?feat(xlsxread):=20stage=203=20writer=20?= =?UTF-8?q?=E2=80=94=20SQLite=20DDL,=20INSERT=20OR=20REPLACE,=20VACUUM,=20?= =?UTF-8?q?stats=20output?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- .../xlsxread/configs/thptqg2017-data-old.toml | 75 +++++++++ .../configs/thptqg2017-data-old2.toml | 76 +++++++++ .../xlsxread/configs/thptqg2017-data.toml | 75 +++++++++ 2017/tools/xlsxread/src/writer.rs | 150 ++++++++++++++++++ 4 files changed, 376 insertions(+) create mode 100644 2017/tools/xlsxread/configs/thptqg2017-data-old.toml create mode 100644 2017/tools/xlsxread/configs/thptqg2017-data-old2.toml create mode 100644 2017/tools/xlsxread/configs/thptqg2017-data.toml create mode 100644 2017/tools/xlsxread/src/writer.rs diff --git a/2017/tools/xlsxread/configs/thptqg2017-data-old.toml b/2017/tools/xlsxread/configs/thptqg2017-data-old.toml new file mode 100644 index 0000000..03d51be --- /dev/null +++ b/2017/tools/xlsxread/configs/thptqg2017-data-old.toml @@ -0,0 +1,75 @@ +# Config for data-old/ — 63 .xlsx files (pre-baotintuc refresh) +# Sheet mode: "first" — single-sheet workbooks, never hit 65k row cap +# SBD validation: require ^\d+$ (build-database-old.js:55 guard) +# Blank row strip: off (no explicit blank-skip in build-database-old.js) + +[reader] +sheet_mode = "first" +strip_blank_rows = false + +[columns] +ho_ten = 0 +ngay_sinh = 1 +so_bao_danh = 2 +diem_thi = 3 + +[validation] +require_numeric_sbd = true +require_nonempty_name = true +require_nonempty_sbd = true + +[header] +tokens = ["HO_TEN", "HỌ TÊN", "STT"] + +[schema] +ddl = """ +CREATE TABLE student ( + so_bao_danh TEXT PRIMARY KEY, + ho_ten TEXT NOT NULL, + ho_ten_ascii TEXT NOT NULL, + ngay_sinh TEXT, + toan REAL, + ngu_van REAL, + vat_ly REAL, + hoa_hoc REAL, + sinh_hoc REAL, + khtn REAL, + lich_su REAL, + dia_ly REAL, + gdcd REAL, + khxh REAL, + tieng_anh REAL, + tieng_phap REAL, + tieng_nga REAL, + tieng_trung REAL +); +CREATE INDEX idx_ho_ten ON student(ho_ten); +CREATE INDEX idx_ho_ten_ascii ON student(ho_ten_ascii); +""" + +[scores] +toan = 'Toán:\s*(\d+(?:\.\d+)?)' +ngu_van = 'Ngữ văn:\s*(\d+(?:\.\d+)?)' +vat_ly = 'Vật lí:\s*(\d+(?:\.\d+)?)' +hoa_hoc = 'Hóa học:\s*(\d+(?:\.\d+)?)' +sinh_hoc = 'Sinh học:\s*(\d+(?:\.\d+)?)' +khtn = 'KHTN:\s*(\d+(?:\.\d+)?)' +lich_su = 'Lịch sử:\s*(\d+(?:\.\d+)?)' +dia_ly = 'Địa lí:\s*(\d+(?:\.\d+)?)' +gdcd = 'GDCD:\s*(\d+(?:\.\d+)?)' +khxh = 'KHXH:\s*(\d+(?:\.\d+)?)' +tieng_anh = 'Tiếng Anh:\s*(\d+(?:\.\d+)?)' +tieng_phap = 'Tiếng Pháp:\s*(\d+(?:\.\d+)?)' +tieng_nga = 'Tiếng Nga:\s*(\d+(?:\.\d+)?)' +tieng_trung = 'Tiếng Trung:\s*(\d+(?:\.\d+)?)' + +[insert] +sql = """ +INSERT OR REPLACE INTO student + (so_bao_danh, ho_ten, ho_ten_ascii, ngay_sinh, + toan, ngu_van, vat_ly, hoa_hoc, sinh_hoc, khtn, + lich_su, dia_ly, gdcd, khxh, + tieng_anh, tieng_phap, tieng_nga, tieng_trung) +VALUES + (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) +""" diff --git a/2017/tools/xlsxread/configs/thptqg2017-data-old2.toml b/2017/tools/xlsxread/configs/thptqg2017-data-old2.toml new file mode 100644 index 0000000..2c3f52c --- /dev/null +++ b/2017/tools/xlsxread/configs/thptqg2017-data-old2.toml @@ -0,0 +1,76 @@ +# Config for data-old2/ — 54 .xlsx files (corrected-export set) +# Sheet mode: "all" — HCM (24.HCM_UTLQ.xlsx) overflows into Sheet2 (+6,446 rows) +# SBD validation: require ^\d+$ (build-database-old2.js:57 guard) +# Blank row strip: true — skip fully blank rows BEFORE counting sourceRows +# (build-database-old2.js:50-51: blank row check before sourceRows++) + +[reader] +sheet_mode = "all" +strip_blank_rows = true + +[columns] +ho_ten = 0 +ngay_sinh = 1 +so_bao_danh = 2 +diem_thi = 3 + +[validation] +require_numeric_sbd = true +require_nonempty_name = true +require_nonempty_sbd = true + +[header] +tokens = ["HO_TEN", "HỌ TÊN", "STT"] + +[schema] +ddl = """ +CREATE TABLE student ( + so_bao_danh TEXT PRIMARY KEY, + ho_ten TEXT NOT NULL, + ho_ten_ascii TEXT NOT NULL, + ngay_sinh TEXT, + toan REAL, + ngu_van REAL, + vat_ly REAL, + hoa_hoc REAL, + sinh_hoc REAL, + khtn REAL, + lich_su REAL, + dia_ly REAL, + gdcd REAL, + khxh REAL, + tieng_anh REAL, + tieng_phap REAL, + tieng_nga REAL, + tieng_trung REAL +); +CREATE INDEX idx_ho_ten ON student(ho_ten); +CREATE INDEX idx_ho_ten_ascii ON student(ho_ten_ascii); +""" + +[scores] +toan = 'Toán:\s*(\d+(?:\.\d+)?)' +ngu_van = 'Ngữ văn:\s*(\d+(?:\.\d+)?)' +vat_ly = 'Vật lí:\s*(\d+(?:\.\d+)?)' +hoa_hoc = 'Hóa học:\s*(\d+(?:\.\d+)?)' +sinh_hoc = 'Sinh học:\s*(\d+(?:\.\d+)?)' +khtn = 'KHTN:\s*(\d+(?:\.\d+)?)' +lich_su = 'Lịch sử:\s*(\d+(?:\.\d+)?)' +dia_ly = 'Địa lí:\s*(\d+(?:\.\d+)?)' +gdcd = 'GDCD:\s*(\d+(?:\.\d+)?)' +khxh = 'KHXH:\s*(\d+(?:\.\d+)?)' +tieng_anh = 'Tiếng Anh:\s*(\d+(?:\.\d+)?)' +tieng_phap = 'Tiếng Pháp:\s*(\d+(?:\.\d+)?)' +tieng_nga = 'Tiếng Nga:\s*(\d+(?:\.\d+)?)' +tieng_trung = 'Tiếng Trung:\s*(\d+(?:\.\d+)?)' + +[insert] +sql = """ +INSERT OR REPLACE INTO student + (so_bao_danh, ho_ten, ho_ten_ascii, ngay_sinh, + toan, ngu_van, vat_ly, hoa_hoc, sinh_hoc, khtn, + lich_su, dia_ly, gdcd, khxh, + tieng_anh, tieng_phap, tieng_nga, tieng_trung) +VALUES + (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) +""" diff --git a/2017/tools/xlsxread/configs/thptqg2017-data.toml b/2017/tools/xlsxread/configs/thptqg2017-data.toml new file mode 100644 index 0000000..c879f7b --- /dev/null +++ b/2017/tools/xlsxread/configs/thptqg2017-data.toml @@ -0,0 +1,75 @@ +# Config for data/ — 63 .xls files from baotintuc.vn +# Sheet mode: "all" because Hà Nội and HCM overflow into Sheet2 (65k row cap) +# SBD validation: no numeric guard (build-database.js does not apply ^\d+$) +# Blank row strip: off + +[reader] +sheet_mode = "all" +strip_blank_rows = false + +[columns] +ho_ten = 0 +ngay_sinh = 1 +so_bao_danh = 2 +diem_thi = 3 + +[validation] +require_numeric_sbd = false +require_nonempty_name = true +require_nonempty_sbd = true + +[header] +tokens = ["HO_TEN", "HỌ TÊN", "STT"] + +[schema] +ddl = """ +CREATE TABLE student ( + so_bao_danh TEXT PRIMARY KEY, + ho_ten TEXT NOT NULL, + ho_ten_ascii TEXT NOT NULL, + ngay_sinh TEXT, + toan REAL, + ngu_van REAL, + vat_ly REAL, + hoa_hoc REAL, + sinh_hoc REAL, + khtn REAL, + lich_su REAL, + dia_ly REAL, + gdcd REAL, + khxh REAL, + tieng_anh REAL, + tieng_phap REAL, + tieng_nga REAL, + tieng_trung REAL +); +CREATE INDEX idx_ho_ten ON student(ho_ten); +CREATE INDEX idx_ho_ten_ascii ON student(ho_ten_ascii); +""" + +[scores] +toan = 'Toán:\s*(\d+(?:\.\d+)?)' +ngu_van = 'Ngữ văn:\s*(\d+(?:\.\d+)?)' +vat_ly = 'Vật lí:\s*(\d+(?:\.\d+)?)' +hoa_hoc = 'Hóa học:\s*(\d+(?:\.\d+)?)' +sinh_hoc = 'Sinh học:\s*(\d+(?:\.\d+)?)' +khtn = 'KHTN:\s*(\d+(?:\.\d+)?)' +lich_su = 'Lịch sử:\s*(\d+(?:\.\d+)?)' +dia_ly = 'Địa lí:\s*(\d+(?:\.\d+)?)' +gdcd = 'GDCD:\s*(\d+(?:\.\d+)?)' +khxh = 'KHXH:\s*(\d+(?:\.\d+)?)' +tieng_anh = 'Tiếng Anh:\s*(\d+(?:\.\d+)?)' +tieng_phap = 'Tiếng Pháp:\s*(\d+(?:\.\d+)?)' +tieng_nga = 'Tiếng Nga:\s*(\d+(?:\.\d+)?)' +tieng_trung = 'Tiếng Trung:\s*(\d+(?:\.\d+)?)' + +[insert] +sql = """ +INSERT OR REPLACE INTO student + (so_bao_danh, ho_ten, ho_ten_ascii, ngay_sinh, + toan, ngu_van, vat_ly, hoa_hoc, sinh_hoc, khtn, + lich_su, dia_ly, gdcd, khxh, + tieng_anh, tieng_phap, tieng_nga, tieng_trung) +VALUES + (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) +""" diff --git a/2017/tools/xlsxread/src/writer.rs b/2017/tools/xlsxread/src/writer.rs new file mode 100644 index 0000000..c75c4ea --- /dev/null +++ b/2017/tools/xlsxread/src/writer.rs @@ -0,0 +1,150 @@ +/// SQLite writer: DDL setup, batched INSERT OR REPLACE, VACUUM, stats output. +/// +/// Mirrors build-lib.js createDb + the transaction loop in each build-database*.js. +/// Stats output lines match the JS stdout exactly so existing CI log-greps still work. +use std::fs; +use std::path::Path; + +use rusqlite::{params_from_iter, Connection, ToSql}; + +use crate::config::DatasetConfig; +use crate::error::BuildError; +use crate::transform::ParsedRow; + +// --------------------------------------------------------------------------- +// DB initialisation — mirrors build-lib.js createDb (delete + recreate) +// --------------------------------------------------------------------------- + +/// Open (or recreate) the output SQLite database, execute the DDL from config, +/// and return the open connection ready for inserts. +pub fn open_db(db_path: &Path, cfg: &DatasetConfig) -> Result { + // Mirror Node behaviour: delete existing file before creating (build-lib.js:54) + if db_path.exists() { + fs::remove_file(db_path).map_err(|e| BuildError::Io { + path: db_path.display().to_string(), + source: e, + })?; + } + + // Ensure parent directory exists + if let Some(parent) = db_path.parent() { + if !parent.as_os_str().is_empty() { + fs::create_dir_all(parent).map_err(|e| BuildError::Io { + path: parent.display().to_string(), + source: e, + })?; + } + } + + let conn = Connection::open(db_path)?; + conn.execute_batch(&cfg.schema.ddl)?; + Ok(conn) +} + +// --------------------------------------------------------------------------- +// Ordered score field list — canonical INSERT column order from build-lib.js +// --------------------------------------------------------------------------- + +/// Fixed subject column order matching the INSERT statement in every config. +/// NULL is bound for any subject not present in a given row's score map. +pub const SCORE_FIELDS: &[&str] = &[ + "toan", + "ngu_van", + "vat_ly", + "hoa_hoc", + "sinh_hoc", + "khtn", + "lich_su", + "dia_ly", + "gdcd", + "khxh", + "tieng_anh", + "tieng_phap", + "tieng_nga", + "tieng_trung", +]; + +// --------------------------------------------------------------------------- +// Insert a single parsed row inside an active transaction +// --------------------------------------------------------------------------- + +/// Bind all fields from `row` into the prepared statement and execute it. +/// `score_fields` should be the ordered list of subject columns the INSERT expects. +pub fn insert_row( + conn: &Connection, + sql: &str, + row: &ParsedRow, + score_fields: &[&str], +) -> Result<(), BuildError> { + // Build positional params: so_bao_danh, ho_ten, ho_ten_ascii, ngay_sinh, + let mut params: Vec> = Vec::with_capacity(4 + score_fields.len()); + params.push(Box::new(row.so_bao_danh.clone())); + params.push(Box::new(row.ho_ten.clone())); + params.push(Box::new(row.ho_ten_ascii.clone())); + params.push(Box::new(row.ngay_sinh.clone())); + + for field in score_fields { + let val: Option = row.scores.get(*field).copied(); + params.push(Box::new(val)); + } + + conn.execute(sql, params_from_iter(params.iter().map(|p| p.as_ref())))?; + Ok(()) +} + +// --------------------------------------------------------------------------- +// Post-build: VACUUM + stats output +// --------------------------------------------------------------------------- + +/// Run VACUUM and print statistics lines that mirror the Node scripts' stdout. +/// The exact prefix tokens ("Source data rows", "DB rows", "Size:") are preserved +/// so any log-grep in the deploy pipeline keeps working. +#[allow(clippy::too_many_arguments)] +pub fn finish_db( + conn: &Connection, + db_path: &Path, + source_rows: u64, + skipped: u64, + errors: u64, + dataset_label: &str, // e.g. "data/" or "data-old2/" + _file_count: usize, + is_old2: bool, // data-old2 uses different label for the skipped line +) -> Result<(), BuildError> { + conn.execute_batch("VACUUM")?; + + let db_count: i64 = conn.query_row("SELECT COUNT(*) FROM student", [], |row| row.get(0))?; + + let insertable = source_rows - skipped; + + // Mirror exact JS stdout format for each dataset variant + println!(); + if is_old2 { + println!("Source non-blank data rows: {source_rows}"); + println!(" skipped (empty/non-numeric SBD): {skipped}"); + } else { + println!("Source data rows (post-header): {source_rows}"); + if dataset_label.contains("old") { + println!(" skipped (empty/non-numeric SBD): {skipped}"); + } else { + println!(" skipped (empty/invalid): {skipped}"); + } + } + println!(" insertable: {insertable}"); + println!(" insert errors: {errors}"); + println!("DB rows (distinct SBD): {db_count}"); + + // Audit gap comment (mirrors build-database.js:80-83 for data/ only) + if !dataset_label.contains("old") && errors == 0 { + let gap = insertable as i64 - db_count; + if gap == 0 { + println!("Audit: OK — every source row made it in."); + } else { + println!("Audit: {gap} row(s) collapsed (duplicate SBDs overwriting)."); + } + } + + let sz = fs::metadata(db_path).map(|m| m.len()).unwrap_or(0); + println!("Size: {:.1} MB", sz as f64 / 1024.0 / 1024.0); + + Ok(()) +}