feat(xlsxread): stage 3 writer — SQLite DDL, INSERT OR REPLACE, VACUUM, stats output

This commit is contained in:
tiennm99 committed 2026-05-19 15:10:03 +07:00
1 parent ceaf8454ed
commit 76f7d84980
4 files changed
+376

No files matched your search

@@ -0,0 +1,75 @@
# Config for data-old/ — 63 .xlsx files (pre-baotintuc refresh)
# Sheet mode: "first" — single-sheet workbooks, never hit 65k row cap
# SBD validation: require ^\d+$ (build-database-old.js:55 guard)
# Blank row strip: off (no explicit blank-skip in build-database-old.js)
[reader]
sheet_mode = "first"
strip_blank_rows = false
[columns]
ho_ten = 0
ngay_sinh = 1
so_bao_danh = 2
diem_thi = 3
[validation]
require_numeric_sbd = true
require_nonempty_name = true
require_nonempty_sbd = true
[header]
tokens = ["HO_TEN", "HỌ TÊN", "STT"]
[schema]
ddl = """
CREATE TABLE student (
so_bao_danh TEXT PRIMARY KEY,
ho_ten TEXT NOT NULL,
ho_ten_ascii TEXT NOT NULL,
ngay_sinh TEXT,
toan REAL,
ngu_van REAL,
vat_ly REAL,
hoa_hoc REAL,
sinh_hoc REAL,
khtn REAL,
lich_su REAL,
dia_ly REAL,
gdcd REAL,
khxh REAL,
tieng_anh REAL,
tieng_phap REAL,
tieng_nga REAL,
tieng_trung REAL
);
CREATE INDEX idx_ho_ten ON student(ho_ten);
CREATE INDEX idx_ho_ten_ascii ON student(ho_ten_ascii);
"""
[scores]
toan = 'Toán:\s*(\d+(?:\.\d+)?)'
ngu_van = 'Ngữ văn:\s*(\d+(?:\.\d+)?)'
vat_ly = 'Vật lí:\s*(\d+(?:\.\d+)?)'
hoa_hoc = 'Hóa học:\s*(\d+(?:\.\d+)?)'
sinh_hoc = 'Sinh học:\s*(\d+(?:\.\d+)?)'
khtn = 'KHTN:\s*(\d+(?:\.\d+)?)'
lich_su = 'Lịch sử:\s*(\d+(?:\.\d+)?)'
dia_ly = 'Địa lí:\s*(\d+(?:\.\d+)?)'
gdcd = 'GDCD:\s*(\d+(?:\.\d+)?)'
khxh = 'KHXH:\s*(\d+(?:\.\d+)?)'
tieng_anh = 'Tiếng Anh:\s*(\d+(?:\.\d+)?)'
tieng_phap = 'Tiếng Pháp:\s*(\d+(?:\.\d+)?)'
tieng_nga = 'Tiếng Nga:\s*(\d+(?:\.\d+)?)'
tieng_trung = 'Tiếng Trung:\s*(\d+(?:\.\d+)?)'
[insert]
sql = """
INSERT OR REPLACE INTO student
(so_bao_danh, ho_ten, ho_ten_ascii, ngay_sinh,
toan, ngu_van, vat_ly, hoa_hoc, sinh_hoc, khtn,
lich_su, dia_ly, gdcd, khxh,
tieng_anh, tieng_phap, tieng_nga, tieng_trung)
VALUES
(?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
"""
@@ -0,0 +1,76 @@
# Config for data-old2/ — 54 .xlsx files (corrected-export set)
# Sheet mode: "all" — HCM (24.HCM_UTLQ.xlsx) overflows into Sheet2 (+6,446 rows)
# SBD validation: require ^\d+$ (build-database-old2.js:57 guard)
# Blank row strip: true — skip fully blank rows BEFORE counting sourceRows
# (build-database-old2.js:50-51: blank row check before sourceRows++)
[reader]
sheet_mode = "all"
strip_blank_rows = true
[columns]
ho_ten = 0
ngay_sinh = 1
so_bao_danh = 2
diem_thi = 3
[validation]
require_numeric_sbd = true
require_nonempty_name = true
require_nonempty_sbd = true
[header]
tokens = ["HO_TEN", "HỌ TÊN", "STT"]
[schema]
ddl = """
CREATE TABLE student (
so_bao_danh TEXT PRIMARY KEY,
ho_ten TEXT NOT NULL,
ho_ten_ascii TEXT NOT NULL,
ngay_sinh TEXT,
toan REAL,
ngu_van REAL,
vat_ly REAL,
hoa_hoc REAL,
sinh_hoc REAL,
khtn REAL,
lich_su REAL,
dia_ly REAL,
gdcd REAL,
khxh REAL,
tieng_anh REAL,
tieng_phap REAL,
tieng_nga REAL,
tieng_trung REAL
);
CREATE INDEX idx_ho_ten ON student(ho_ten);
CREATE INDEX idx_ho_ten_ascii ON student(ho_ten_ascii);
"""
[scores]
toan = 'Toán:\s*(\d+(?:\.\d+)?)'
ngu_van = 'Ngữ văn:\s*(\d+(?:\.\d+)?)'
vat_ly = 'Vật lí:\s*(\d+(?:\.\d+)?)'
hoa_hoc = 'Hóa học:\s*(\d+(?:\.\d+)?)'
sinh_hoc = 'Sinh học:\s*(\d+(?:\.\d+)?)'
khtn = 'KHTN:\s*(\d+(?:\.\d+)?)'
lich_su = 'Lịch sử:\s*(\d+(?:\.\d+)?)'
dia_ly = 'Địa lí:\s*(\d+(?:\.\d+)?)'
gdcd = 'GDCD:\s*(\d+(?:\.\d+)?)'
khxh = 'KHXH:\s*(\d+(?:\.\d+)?)'
tieng_anh = 'Tiếng Anh:\s*(\d+(?:\.\d+)?)'
tieng_phap = 'Tiếng Pháp:\s*(\d+(?:\.\d+)?)'
tieng_nga = 'Tiếng Nga:\s*(\d+(?:\.\d+)?)'
tieng_trung = 'Tiếng Trung:\s*(\d+(?:\.\d+)?)'
[insert]
sql = """
INSERT OR REPLACE INTO student
(so_bao_danh, ho_ten, ho_ten_ascii, ngay_sinh,
toan, ngu_van, vat_ly, hoa_hoc, sinh_hoc, khtn,
lich_su, dia_ly, gdcd, khxh,
tieng_anh, tieng_phap, tieng_nga, tieng_trung)
VALUES
(?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
"""
@@ -0,0 +1,75 @@
# Config for data/ — 63 .xls files from baotintuc.vn
# Sheet mode: "all" because Hà Nội and HCM overflow into Sheet2 (65k row cap)
# SBD validation: no numeric guard (build-database.js does not apply ^\d+$)
# Blank row strip: off
[reader]
sheet_mode = "all"
strip_blank_rows = false
[columns]
ho_ten = 0
ngay_sinh = 1
so_bao_danh = 2
diem_thi = 3
[validation]
require_numeric_sbd = false
require_nonempty_name = true
require_nonempty_sbd = true
[header]
tokens = ["HO_TEN", "HỌ TÊN", "STT"]
[schema]
ddl = """
CREATE TABLE student (
so_bao_danh TEXT PRIMARY KEY,
ho_ten TEXT NOT NULL,
ho_ten_ascii TEXT NOT NULL,
ngay_sinh TEXT,
toan REAL,
ngu_van REAL,
vat_ly REAL,
hoa_hoc REAL,
sinh_hoc REAL,
khtn REAL,
lich_su REAL,
dia_ly REAL,
gdcd REAL,
khxh REAL,
tieng_anh REAL,
tieng_phap REAL,
tieng_nga REAL,
tieng_trung REAL
);
CREATE INDEX idx_ho_ten ON student(ho_ten);
CREATE INDEX idx_ho_ten_ascii ON student(ho_ten_ascii);
"""
[scores]
toan = 'Toán:\s*(\d+(?:\.\d+)?)'
ngu_van = 'Ngữ văn:\s*(\d+(?:\.\d+)?)'
vat_ly = 'Vật lí:\s*(\d+(?:\.\d+)?)'
hoa_hoc = 'Hóa học:\s*(\d+(?:\.\d+)?)'
sinh_hoc = 'Sinh học:\s*(\d+(?:\.\d+)?)'
khtn = 'KHTN:\s*(\d+(?:\.\d+)?)'
lich_su = 'Lịch sử:\s*(\d+(?:\.\d+)?)'
dia_ly = 'Địa lí:\s*(\d+(?:\.\d+)?)'
gdcd = 'GDCD:\s*(\d+(?:\.\d+)?)'
khxh = 'KHXH:\s*(\d+(?:\.\d+)?)'
tieng_anh = 'Tiếng Anh:\s*(\d+(?:\.\d+)?)'
tieng_phap = 'Tiếng Pháp:\s*(\d+(?:\.\d+)?)'
tieng_nga = 'Tiếng Nga:\s*(\d+(?:\.\d+)?)'
tieng_trung = 'Tiếng Trung:\s*(\d+(?:\.\d+)?)'
[insert]
sql = """
INSERT OR REPLACE INTO student
(so_bao_danh, ho_ten, ho_ten_ascii, ngay_sinh,
toan, ngu_van, vat_ly, hoa_hoc, sinh_hoc, khtn,
lich_su, dia_ly, gdcd, khxh,
tieng_anh, tieng_phap, tieng_nga, tieng_trung)
VALUES
(?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
"""
+150
View File
@@ -0,0 +1,150 @@
/// SQLite writer: DDL setup, batched INSERT OR REPLACE, VACUUM, stats output.
///
/// Mirrors build-lib.js createDb + the transaction loop in each build-database*.js.
/// Stats output lines match the JS stdout exactly so existing CI log-greps still work.
use std::fs;
use std::path::Path;
use rusqlite::{params_from_iter, Connection, ToSql};
use crate::config::DatasetConfig;
use crate::error::BuildError;
use crate::transform::ParsedRow;
// ---------------------------------------------------------------------------
// DB initialisation — mirrors build-lib.js createDb (delete + recreate)
// ---------------------------------------------------------------------------
/// Open (or recreate) the output SQLite database, execute the DDL from config,
/// and return the open connection ready for inserts.
pub fn open_db(db_path: &Path, cfg: &DatasetConfig) -> Result<Connection, BuildError> {
// Mirror Node behaviour: delete existing file before creating (build-lib.js:54)
if db_path.exists() {
fs::remove_file(db_path).map_err(|e| BuildError::Io {
path: db_path.display().to_string(),
source: e,
})?;
}
// Ensure parent directory exists
if let Some(parent) = db_path.parent() {
if !parent.as_os_str().is_empty() {
fs::create_dir_all(parent).map_err(|e| BuildError::Io {
path: parent.display().to_string(),
source: e,
})?;
}
}
let conn = Connection::open(db_path)?;
conn.execute_batch(&cfg.schema.ddl)?;
Ok(conn)
}
// ---------------------------------------------------------------------------
// Ordered score field list — canonical INSERT column order from build-lib.js
// ---------------------------------------------------------------------------
/// Fixed subject column order matching the INSERT statement in every config.
/// NULL is bound for any subject not present in a given row's score map.
pub const SCORE_FIELDS: &[&str] = &[
"toan",
"ngu_van",
"vat_ly",
"hoa_hoc",
"sinh_hoc",
"khtn",
"lich_su",
"dia_ly",
"gdcd",
"khxh",
"tieng_anh",
"tieng_phap",
"tieng_nga",
"tieng_trung",
];
// ---------------------------------------------------------------------------
// Insert a single parsed row inside an active transaction
// ---------------------------------------------------------------------------
/// Bind all fields from `row` into the prepared statement and execute it.
/// `score_fields` should be the ordered list of subject columns the INSERT expects.
pub fn insert_row(
conn: &Connection,
sql: &str,
row: &ParsedRow,
score_fields: &[&str],
) -> Result<(), BuildError> {
// Build positional params: so_bao_danh, ho_ten, ho_ten_ascii, ngay_sinh, <scores...>
let mut params: Vec<Box<dyn ToSql>> = Vec::with_capacity(4 + score_fields.len());
params.push(Box::new(row.so_bao_danh.clone()));
params.push(Box::new(row.ho_ten.clone()));
params.push(Box::new(row.ho_ten_ascii.clone()));
params.push(Box::new(row.ngay_sinh.clone()));
for field in score_fields {
let val: Option<f64> = row.scores.get(*field).copied();
params.push(Box::new(val));
}
conn.execute(sql, params_from_iter(params.iter().map(|p| p.as_ref())))?;
Ok(())
}
// ---------------------------------------------------------------------------
// Post-build: VACUUM + stats output
// ---------------------------------------------------------------------------
/// Run VACUUM and print statistics lines that mirror the Node scripts' stdout.
/// The exact prefix tokens ("Source data rows", "DB rows", "Size:") are preserved
/// so any log-grep in the deploy pipeline keeps working.
#[allow(clippy::too_many_arguments)]
pub fn finish_db(
conn: &Connection,
db_path: &Path,
source_rows: u64,
skipped: u64,
errors: u64,
dataset_label: &str, // e.g. "data/" or "data-old2/"
_file_count: usize,
is_old2: bool, // data-old2 uses different label for the skipped line
) -> Result<(), BuildError> {
conn.execute_batch("VACUUM")?;
let db_count: i64 = conn.query_row("SELECT COUNT(*) FROM student", [], |row| row.get(0))?;
let insertable = source_rows - skipped;
// Mirror exact JS stdout format for each dataset variant
println!();
if is_old2 {
println!("Source non-blank data rows: {source_rows}");
println!(" skipped (empty/non-numeric SBD): {skipped}");
} else {
println!("Source data rows (post-header): {source_rows}");
if dataset_label.contains("old") {
println!(" skipped (empty/non-numeric SBD): {skipped}");
} else {
println!(" skipped (empty/invalid): {skipped}");
}
}
println!(" insertable: {insertable}");
println!(" insert errors: {errors}");
println!("DB rows (distinct SBD): {db_count}");
// Audit gap comment (mirrors build-database.js:80-83 for data/ only)
if !dataset_label.contains("old") && errors == 0 {
let gap = insertable as i64 - db_count;
if gap == 0 {
println!("Audit: OK — every source row made it in.");
} else {
println!("Audit: {gap} row(s) collapsed (duplicate SBDs overwriting).");
}
}
let sz = fs::metadata(db_path).map(|m| m.len()).unwrap_or(0);
println!("Size: {:.1} MB", sz as f64 / 1024.0 / 1024.0);
Ok(())
}