diff --git a/2016/tools/xlsxread/configs/thptqg2016-data.toml b/2016/tools/xlsxread/configs/thptqg2016-data.toml index 646ad68..6bde781 100644 --- a/2016/tools/xlsxread/configs/thptqg2016-data.toml +++ b/2016/tools/xlsxread/configs/thptqg2016-data.toml @@ -1,4 +1,4 @@ -# Config for thptqg2016 data/ — 4 .xls + 115 .xlsx mixed files. +# thptqg2016 data/ — 4 .xls + 115 .xlsx mixed files. # # Three column layouts exist across the 119 files; the binary selects the # right one per-file at runtime via format_detection = "thptqg2016": @@ -7,15 +7,13 @@ # mapped header SOBAODANH|SBD + DIEM_THI — most provinces # default no header; positional 6-col layout — remaining files # -# Schema differences from thptqg2017: -# + ten_cum_thi TEXT (exam-cluster name, TEN_CUMTHI column) -# + gioi_tinh TEXT (gender: "Nam"/"Nữ", GIOI_TINH column) -# + tieng_duc REAL (German) -# + tieng_nhat REAL (Japanese) -# - khtn, khxh, tieng_nga (not in 2016 dataset) +# This is the only dataset that populates ten_cum_thi and gioi_tinh, and the +# only one whose files carry Tiếng Đức / Tiếng Nhật scores. # # sheet_mode = "all": several provinces overflow into Sheet2 (65k Excel row cap). # strip_blank_rows = false: no blank-row anomaly observed in this dataset. +# +# Table shape, INSERT and subject regexes are canonical — see src/schema.rs. format_detection = "thptqg2016" @@ -32,55 +30,3 @@ require_nonempty_sbd = true # Tokens that identify a header row by first-cell content (uppercased). # Covers both SOBAODANH-style and SBD-style headers. tokens = ["SOBAODANH", "SBD", "HO_TEN", "HOTEN", "HỌ TÊN", "STT"] - -[schema] -ddl = """ -CREATE TABLE student ( - so_bao_danh TEXT PRIMARY KEY, - ho_ten TEXT NOT NULL, - ho_ten_ascii TEXT NOT NULL, - ngay_sinh TEXT, - ten_cum_thi TEXT, - gioi_tinh TEXT, - toan REAL, - ngu_van REAL, - vat_ly REAL, - hoa_hoc REAL, - sinh_hoc REAL, - lich_su REAL, - dia_ly REAL, - tieng_anh REAL, - tieng_phap REAL, - tieng_duc REAL, - tieng_nhat REAL, - tieng_trung REAL -); -CREATE INDEX idx_ho_ten ON student(ho_ten); -CREATE INDEX idx_ho_ten_ascii ON student(ho_ten_ascii); -CREATE INDEX idx_ten_cum_thi ON student(ten_cum_thi); -""" - -[scores] -toan = 'Toán:\s*(\d+(?:\.\d+)?)' -ngu_van = 'Ngữ văn:\s*(\d+(?:\.\d+)?)' -vat_ly = 'Vật lí:\s*(\d+(?:\.\d+)?)' -hoa_hoc = 'Hóa học:\s*(\d+(?:\.\d+)?)' -sinh_hoc = 'Sinh học:\s*(\d+(?:\.\d+)?)' -lich_su = 'Lịch sử:\s*(\d+(?:\.\d+)?)' -dia_ly = 'Địa lí:\s*(\d+(?:\.\d+)?)' -tieng_anh = 'Tiếng Anh:\s*(\d+(?:\.\d+)?)' -tieng_phap = 'Tiếng Pháp:\s*(\d+(?:\.\d+)?)' -tieng_duc = 'Tiếng Đức:\s*(\d+(?:\.\d+)?)' -tieng_nhat = 'Tiếng Nhật:\s*(\d+(?:\.\d+)?)' -tieng_trung = 'Tiếng Trung:\s*(\d+(?:\.\d+)?)' - -[insert] -sql = """ -INSERT OR REPLACE INTO student - (so_bao_danh, ho_ten, ho_ten_ascii, ngay_sinh, ten_cum_thi, gioi_tinh, - toan, ngu_van, vat_ly, hoa_hoc, sinh_hoc, - lich_su, dia_ly, - tieng_anh, tieng_phap, tieng_duc, tieng_nhat, tieng_trung) -VALUES - (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) -""" diff --git a/2016/tools/xlsxread/configs/thptqg2017-data-old.toml b/2016/tools/xlsxread/configs/thptqg2017-data-old.toml index 34d59c0..f21f8aa 100644 --- a/2016/tools/xlsxread/configs/thptqg2017-data-old.toml +++ b/2016/tools/xlsxread/configs/thptqg2017-data-old.toml @@ -1,8 +1,10 @@ -# Test-only config used by golden integration tests (data-old variant). -# Same column layout as thptqg2017-data.toml but with: -# - sheet_mode = "first" (reads only first sheet) -# - require_numeric_sbd = true (rejects non-digit SBDs) -# Schema and INSERT match SCORE_FIELDS_2017 (14 score cols). +# thptqg2017 data-old/ — 63 .xlsx files (pre-baotintuc refresh). +# +# sheet_mode = "first": single-sheet workbooks, never hit the 65k row cap. +# SBD validation: require ^\d+$ (build-database-old.js:55 guard). +# strip_blank_rows = false: no explicit blank-skip in build-database-old.js. +# +# Table shape, INSERT and subject regexes are canonical — see src/schema.rs. [reader] sheet_mode = "first" @@ -21,56 +23,3 @@ require_nonempty_sbd = true [header] tokens = ["HO_TEN", "HỌ TÊN", "STT"] - -[schema] -ddl = """ -CREATE TABLE student ( - so_bao_danh TEXT PRIMARY KEY, - ho_ten TEXT NOT NULL, - ho_ten_ascii TEXT NOT NULL, - ngay_sinh TEXT, - toan REAL, - ngu_van REAL, - vat_ly REAL, - hoa_hoc REAL, - sinh_hoc REAL, - khtn REAL, - lich_su REAL, - dia_ly REAL, - gdcd REAL, - khxh REAL, - tieng_anh REAL, - tieng_phap REAL, - tieng_nga REAL, - tieng_trung REAL -); -CREATE INDEX idx_ho_ten ON student(ho_ten); -CREATE INDEX idx_ho_ten_ascii ON student(ho_ten_ascii); -""" - -[scores] -toan = 'Toán:\s*(\d+(?:\.\d+)?)' -ngu_van = 'Ngữ văn:\s*(\d+(?:\.\d+)?)' -vat_ly = 'Vật lí:\s*(\d+(?:\.\d+)?)' -hoa_hoc = 'Hóa học:\s*(\d+(?:\.\d+)?)' -sinh_hoc = 'Sinh học:\s*(\d+(?:\.\d+)?)' -khtn = 'KHTN:\s*(\d+(?:\.\d+)?)' -lich_su = 'Lịch sử:\s*(\d+(?:\.\d+)?)' -dia_ly = 'Địa lí:\s*(\d+(?:\.\d+)?)' -gdcd = 'GDCD:\s*(\d+(?:\.\d+)?)' -khxh = 'KHXH:\s*(\d+(?:\.\d+)?)' -tieng_anh = 'Tiếng Anh:\s*(\d+(?:\.\d+)?)' -tieng_phap = 'Tiếng Pháp:\s*(\d+(?:\.\d+)?)' -tieng_nga = 'Tiếng Nga:\s*(\d+(?:\.\d+)?)' -tieng_trung = 'Tiếng Trung:\s*(\d+(?:\.\d+)?)' - -[insert] -sql = """ -INSERT OR REPLACE INTO student - (so_bao_danh, ho_ten, ho_ten_ascii, ngay_sinh, - toan, ngu_van, vat_ly, hoa_hoc, sinh_hoc, khtn, - lich_su, dia_ly, gdcd, khxh, - tieng_anh, tieng_phap, tieng_nga, tieng_trung) -VALUES - (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) -""" diff --git a/2016/tools/xlsxread/configs/thptqg2017-data-old2.toml b/2016/tools/xlsxread/configs/thptqg2017-data-old2.toml new file mode 100644 index 0000000..c4fd6b7 --- /dev/null +++ b/2016/tools/xlsxread/configs/thptqg2017-data-old2.toml @@ -0,0 +1,26 @@ +# thptqg2017 data-old2/ — 54 .xlsx files (corrected-export set). +# +# sheet_mode = "all": 24.HCM_UTLQ.xlsx overflows into Sheet2 (+6,446 rows). +# SBD validation: require ^\d+$ (build-database-old2.js:57 guard). +# strip_blank_rows = true: skip fully blank rows BEFORE counting sourceRows +# (build-database-old2.js:50-51 checks blank before sourceRows++). +# +# Table shape, INSERT and subject regexes are canonical — see src/schema.rs. + +[reader] +sheet_mode = "all" +strip_blank_rows = true + +[columns] +ho_ten = 0 +ngay_sinh = 1 +so_bao_danh = 2 +diem_thi = 3 + +[validation] +require_numeric_sbd = true +require_nonempty_name = true +require_nonempty_sbd = true + +[header] +tokens = ["HO_TEN", "HỌ TÊN", "STT"] diff --git a/2016/tools/xlsxread/configs/thptqg2017-data.toml b/2016/tools/xlsxread/configs/thptqg2017-data.toml index 892fc6f..99dccbb 100644 --- a/2016/tools/xlsxread/configs/thptqg2017-data.toml +++ b/2016/tools/xlsxread/configs/thptqg2017-data.toml @@ -1,8 +1,10 @@ -# Test-only config used by golden integration tests. -# Matches the fixture row layout produced by tests/golden.rs: write_xlsx() -# header: HO_TEN(0) NGAY_SINH(1) SO_BAO_DANH(2) DIEM_THI(3) -# Schema and INSERT match SCORE_FIELDS_2017 (14 score cols) so run_build_cmd -# in golden.rs can call insert_row(..., SCORE_FIELDS) without modification. +# thptqg2017 data/ — 63 .xls files from baotintuc.vn (current generation). +# +# sheet_mode = "all": Hà Nội and HCM overflow into Sheet2 (65k Excel row cap). +# SBD validation: no numeric guard — build-database.js did not apply ^\d+$. +# strip_blank_rows = false: no blank-row anomaly in this dataset. +# +# Table shape, INSERT and subject regexes are canonical — see src/schema.rs. [reader] sheet_mode = "all" @@ -21,56 +23,3 @@ require_nonempty_sbd = true [header] tokens = ["HO_TEN", "HỌ TÊN", "STT"] - -[schema] -ddl = """ -CREATE TABLE student ( - so_bao_danh TEXT PRIMARY KEY, - ho_ten TEXT NOT NULL, - ho_ten_ascii TEXT NOT NULL, - ngay_sinh TEXT, - toan REAL, - ngu_van REAL, - vat_ly REAL, - hoa_hoc REAL, - sinh_hoc REAL, - khtn REAL, - lich_su REAL, - dia_ly REAL, - gdcd REAL, - khxh REAL, - tieng_anh REAL, - tieng_phap REAL, - tieng_nga REAL, - tieng_trung REAL -); -CREATE INDEX idx_ho_ten ON student(ho_ten); -CREATE INDEX idx_ho_ten_ascii ON student(ho_ten_ascii); -""" - -[scores] -toan = 'Toán:\s*(\d+(?:\.\d+)?)' -ngu_van = 'Ngữ văn:\s*(\d+(?:\.\d+)?)' -vat_ly = 'Vật lí:\s*(\d+(?:\.\d+)?)' -hoa_hoc = 'Hóa học:\s*(\d+(?:\.\d+)?)' -sinh_hoc = 'Sinh học:\s*(\d+(?:\.\d+)?)' -khtn = 'KHTN:\s*(\d+(?:\.\d+)?)' -lich_su = 'Lịch sử:\s*(\d+(?:\.\d+)?)' -dia_ly = 'Địa lí:\s*(\d+(?:\.\d+)?)' -gdcd = 'GDCD:\s*(\d+(?:\.\d+)?)' -khxh = 'KHXH:\s*(\d+(?:\.\d+)?)' -tieng_anh = 'Tiếng Anh:\s*(\d+(?:\.\d+)?)' -tieng_phap = 'Tiếng Pháp:\s*(\d+(?:\.\d+)?)' -tieng_nga = 'Tiếng Nga:\s*(\d+(?:\.\d+)?)' -tieng_trung = 'Tiếng Trung:\s*(\d+(?:\.\d+)?)' - -[insert] -sql = """ -INSERT OR REPLACE INTO student - (so_bao_danh, ho_ten, ho_ten_ascii, ngay_sinh, - toan, ngu_van, vat_ly, hoa_hoc, sinh_hoc, khtn, - lich_su, dia_ly, gdcd, khxh, - tieng_anh, tieng_phap, tieng_nga, tieng_trung) -VALUES - (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) -""" diff --git a/2016/tools/xlsxread/src/config.rs b/2016/tools/xlsxread/src/config.rs index e2aaa8e..c23e090 100644 --- a/2016/tools/xlsxread/src/config.rs +++ b/2016/tools/xlsxread/src/config.rs @@ -1,4 +1,3 @@ -use std::collections::HashMap; use std::fs; use std::path::Path; @@ -10,7 +9,14 @@ use crate::error::BuildError; // Top-level dataset configuration loaded from a .toml file // --------------------------------------------------------------------------- +/// Per-dataset parse rules. +/// +/// Deliberately carries no SQL. The table shape, the INSERT and the subject +/// regexes are identical for every dataset and live in `crate::schema` — keeping +/// them here meant four copies of the same DDL, which is how the 2016 and 2017 +/// schemas drifted apart. #[derive(Debug, Deserialize, Clone)] +#[serde(deny_unknown_fields)] pub struct DatasetConfig { pub reader: ReaderCfg, /// Fixed column indices. Optional when format_detection handles per-file mapping. @@ -18,10 +24,6 @@ pub struct DatasetConfig { pub columns: Option, pub validation: ValidationCfg, pub header: HeaderCfg, - pub schema: SchemaCfg, - /// field name → regex source string (one entry per scoreable subject) - pub scores: HashMap, - pub insert: InsertCfg, /// When set to "thptqg2016", enables per-file format auto-detection. /// Each file's header row is inspected at runtime to choose the right /// column layout (separate-scores / mapped / default-positional). @@ -68,18 +70,6 @@ pub struct HeaderCfg { pub tokens: Vec, } -#[derive(Debug, Deserialize, Clone)] -pub struct SchemaCfg { - /// DDL executed verbatim before inserts (CREATE TABLE + CREATE INDEX) - pub ddl: String, -} - -#[derive(Debug, Deserialize, Clone)] -pub struct InsertCfg { - /// Parameterised INSERT OR REPLACE SQL using :named_param style - pub sql: String, -} - // --------------------------------------------------------------------------- // Loader // --------------------------------------------------------------------------- @@ -119,16 +109,6 @@ require_nonempty_sbd = true [header] tokens = ["HO_TEN", "HỌ TÊN", "STT"] - -[schema] -ddl = "CREATE TABLE student (so_bao_danh TEXT PRIMARY KEY);" - -[scores] -toan = 'Toán:\s*(\d+(?:\.\d+)?)' -ngu_van = 'Ngữ văn:\s*(\d+(?:\.\d+)?)' - -[insert] -sql = "INSERT OR REPLACE INTO student (so_bao_danh) VALUES (:so_bao_danh)" "#; #[test] @@ -142,11 +122,20 @@ sql = "INSERT OR REPLACE INTO student (so_bao_danh) VALUES (:so_bao_danh)" assert!(!cfg.validation.require_numeric_sbd); assert!(cfg.validation.require_nonempty_name); assert_eq!(cfg.header.tokens.len(), 3); - assert!(cfg.scores.contains_key("toan")); - assert!(cfg.scores.contains_key("ngu_van")); assert!(cfg.format_detection.is_none()); } + /// A config carrying leftover SQL sections must be rejected rather than + /// silently ignored — otherwise a stale [schema] block would look effective + /// while `crate::schema` was actually driving the build. + #[test] + fn config_rejects_leftover_sql_sections() { + let with_ddl = format!( + "{SAMPLE_TOML}\n[schema]\nddl = \"CREATE TABLE student (so_bao_danh TEXT);\"\n" + ); + assert!(toml::from_str::(&with_ddl).is_err()); + } + #[test] fn config_first_sheet_mode() { let toml_str = SAMPLE_TOML.replace(r#"sheet_mode = "all""#, r#"sheet_mode = "first""#); @@ -171,15 +160,6 @@ require_nonempty_sbd = true [header] tokens = ["SBD", "SOBAODANH", "STT"] - -[schema] -ddl = "CREATE TABLE student (so_bao_danh TEXT PRIMARY KEY);" - -[scores] -toan = 'Toán:\s*(\d+(?:\.\d+)?)' - -[insert] -sql = "INSERT OR REPLACE INTO student (so_bao_danh) VALUES (?)" "#; let cfg: DatasetConfig = toml::from_str(toml_str).expect("parse failed"); assert_eq!(cfg.format_detection.as_deref(), Some("thptqg2016")); diff --git a/2016/tools/xlsxread/src/format_detect_2016.rs b/2016/tools/xlsxread/src/format_detect_2016.rs index 36c2bfd..768a88c 100644 --- a/2016/tools/xlsxread/src/format_detect_2016.rs +++ b/2016/tools/xlsxread/src/format_detect_2016.rs @@ -337,20 +337,13 @@ pub fn process_row_2016( #[cfg(test)] mod tests { use super::*; - use std::collections::HashMap; fn s(v: &str) -> Data { Data::String(v.to_string()) } fn make_patterns() -> CompiledPatterns { - let mut map = HashMap::new(); - map.insert("toan".into(), r"Toán:\s*(\d+(?:\.\d+)?)".into()); - map.insert("ngu_van".into(), r"Ngữ văn:\s*(\d+(?:\.\d+)?)".into()); - map.insert("tieng_anh".into(), r"Tiếng Anh:\s*(\d+(?:\.\d+)?)".into()); - map.insert("tieng_duc".into(), r"Tiếng Đức:\s*(\d+(?:\.\d+)?)".into()); - map.insert("tieng_nhat".into(), r"Tiếng Nhật:\s*(\d+(?:\.\d+)?)".into()); - CompiledPatterns::new(&map).unwrap() + CompiledPatterns::new().unwrap() } // --- detect_format: branch 1 — separate-scores --- diff --git a/2016/tools/xlsxread/src/lib.rs b/2016/tools/xlsxread/src/lib.rs index b61560c..ce7f5e4 100644 --- a/2016/tools/xlsxread/src/lib.rs +++ b/2016/tools/xlsxread/src/lib.rs @@ -6,5 +6,6 @@ pub mod config; pub mod error; pub mod format_detect_2016; pub mod reader; +pub mod schema; pub mod transform; pub mod writer; diff --git a/2016/tools/xlsxread/src/main.rs b/2016/tools/xlsxread/src/main.rs index 0736ed7..fd8b342 100644 --- a/2016/tools/xlsxread/src/main.rs +++ b/2016/tools/xlsxread/src/main.rs @@ -25,9 +25,7 @@ use xlsxread::format_detect_2016::{ }; use xlsxread::reader::{is_all_blank, process_file}; use xlsxread::transform::{validate_row, CompiledPatterns, SkipReason}; -use xlsxread::writer::{ - finish_db, insert_row, insert_row_2016, open_db, SCORE_FIELDS, SCORE_FIELDS_2016, -}; +use xlsxread::writer::{finish_db, insert_row, open_db}; fn main() -> Result<()> { let cli = Cli::parse(); @@ -79,7 +77,7 @@ fn run_build_standard( output_path: &Path, ) -> Result<()> { let patterns = - CompiledPatterns::new(&cfg.scores).with_context(|| "Failed to compile score regexes")?; + CompiledPatterns::new().with_context(|| "Failed to compile score regexes")?; let mut files: Vec = std::fs::read_dir(input_dir) .with_context(|| format!("Cannot read input dir: {}", input_dir.display()))? @@ -109,7 +107,7 @@ fn run_build_standard( files.len() ); - let conn = open_db(output_path, cfg) + let conn = open_db(output_path) .with_context(|| format!("Failed to open DB: {}", output_path.display()))?; let mut total_source_rows: u64 = 0; @@ -159,7 +157,7 @@ fn run_build_standard( let parsed = xlsxread::transform::transform_row(raw, cfg, &patterns); - match insert_row(&conn, &cfg.insert.sql, &parsed, SCORE_FIELDS) { + match insert_row(&conn, &parsed) { Ok(()) => file_rows += 1, Err(e) => { file_errors += 1; @@ -216,7 +214,7 @@ fn run_build_2016( output_path: &Path, ) -> Result<()> { let patterns = - CompiledPatterns::new(&cfg.scores).with_context(|| "Failed to compile score regexes")?; + CompiledPatterns::new().with_context(|| "Failed to compile score regexes")?; let mut files: Vec = std::fs::read_dir(input_dir) .with_context(|| format!("Cannot read input dir: {}", input_dir.display()))? @@ -246,7 +244,7 @@ fn run_build_2016( files.len() ); - let conn = open_db(output_path, cfg) + let conn = open_db(output_path) .with_context(|| format!("Failed to open DB: {}", output_path.display()))?; let mut total_source_rows: u64 = 0; @@ -361,7 +359,7 @@ fn process_file_2016( // Row was empty/invalid — skipped (mirrors JS `if (!record) continue`) } Some(parsed) => { - match insert_row_2016(conn, &cfg.insert.sql, &parsed, SCORE_FIELDS_2016) { + match insert_row(conn, &parsed) { Ok(()) => file_rows += 1, Err(e) => { *total_errors += 1; diff --git a/2016/tools/xlsxread/src/schema.rs b/2016/tools/xlsxread/src/schema.rs new file mode 100644 index 0000000..6eaf739 --- /dev/null +++ b/2016/tools/xlsxread/src/schema.rs @@ -0,0 +1,213 @@ +//! Canonical `student` table definition — the single source of truth for the +//! SQL shape of every dataset. +//! +//! Every dataset (2016, 2017, 2017-old, 2017-old2) is written into this same +//! 22-column table. Columns a dataset has no data for bind NULL, which costs +//! ~1 byte per row in SQLite's record header. +//! +//! Column provenance: +//! ten_cum_thi, gioi_tinh, tieng_duc, tieng_nhat → 2016 only +//! khtn, khxh, gdcd, tieng_nga → 2017 datasets only +//! everything else → both +//! +//! Before this module existed the DDL, the INSERT statement and the subject +//! regex table were duplicated across four TOML configs, which is how the 2016 +//! and 2017 schemas drifted apart in the first place. The configs now carry only +//! per-dataset parse rules. +// --------------------------------------------------------------------------- +// DDL +// --------------------------------------------------------------------------- + +/// Executed verbatim after the output database is (re)created. +/// +/// `idx_ten_cum_thi` is partial so it holds zero entries on the three 2017 +/// datasets — where the column is always NULL — while staying fully useful for +/// the 2016 cluster-grouping queries. +pub const DDL: &str = " +CREATE TABLE student ( + so_bao_danh TEXT PRIMARY KEY, + ho_ten TEXT NOT NULL, + ho_ten_ascii TEXT NOT NULL, + ngay_sinh TEXT, + ten_cum_thi TEXT, + gioi_tinh TEXT, + toan REAL, + ngu_van REAL, + vat_ly REAL, + hoa_hoc REAL, + sinh_hoc REAL, + khtn REAL, + lich_su REAL, + dia_ly REAL, + gdcd REAL, + khxh REAL, + tieng_anh REAL, + tieng_phap REAL, + tieng_nga REAL, + tieng_duc REAL, + tieng_nhat REAL, + tieng_trung REAL +); +CREATE INDEX idx_ho_ten ON student(ho_ten); +CREATE INDEX idx_ho_ten_ascii ON student(ho_ten_ascii); +CREATE INDEX idx_ten_cum_thi ON student(ten_cum_thi) WHERE ten_cum_thi IS NOT NULL; +"; + +// --------------------------------------------------------------------------- +// Column order +// --------------------------------------------------------------------------- + +/// Identity columns, in INSERT parameter order. +pub const IDENTITY_FIELDS: &[&str] = &[ + "so_bao_danh", + "ho_ten", + "ho_ten_ascii", + "ngay_sinh", + "ten_cum_thi", + "gioi_tinh", +]; + +/// Subject columns, in INSERT parameter order. Bound as NULL when a row has no +/// score for that subject. +pub const SCORE_FIELDS: &[&str] = &[ + "toan", + "ngu_van", + "vat_ly", + "hoa_hoc", + "sinh_hoc", + "khtn", + "lich_su", + "dia_ly", + "gdcd", + "khxh", + "tieng_anh", + "tieng_phap", + "tieng_nga", + "tieng_duc", + "tieng_nhat", + "tieng_trung", +]; + +/// Total bound parameters per row. +pub const PARAM_COUNT: usize = IDENTITY_FIELDS.len() + SCORE_FIELDS.len(); + +// --------------------------------------------------------------------------- +// INSERT +// --------------------------------------------------------------------------- + +/// Positional INSERT matching IDENTITY_FIELDS followed by SCORE_FIELDS. +/// +/// `OR REPLACE` preserves the pre-existing behaviour where a repeated SBD +/// overwrites the earlier row rather than aborting the transaction. +pub const INSERT_SQL: &str = " +INSERT OR REPLACE INTO student + (so_bao_danh, ho_ten, ho_ten_ascii, ngay_sinh, ten_cum_thi, gioi_tinh, + toan, ngu_van, vat_ly, hoa_hoc, sinh_hoc, khtn, + lich_su, dia_ly, gdcd, khxh, + tieng_anh, tieng_phap, tieng_nga, tieng_duc, tieng_nhat, tieng_trung) +VALUES + (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) +"; + +// --------------------------------------------------------------------------- +// Subject score patterns +// --------------------------------------------------------------------------- + +/// Regex per subject, applied to the DIEM_THI cell text. +/// +/// Every pattern runs against every dataset. A subject absent from a given exam +/// year simply never matches and stays NULL — 2016 source files contain no +/// "KHTN:" or "Tiếng Nga:" tokens, and 2017 files contain no "Tiếng Đức:" or +/// "Tiếng Nhật:". The parity check asserts those counts are exactly zero rather +/// than assuming it. +/// +/// Order here is irrelevant (matching is by name); SCORE_FIELDS fixes the +/// INSERT order. +pub const SCORE_PATTERNS: &[(&str, &str)] = &[ + ("toan", r"Toán:\s*(\d+(?:\.\d+)?)"), + ("ngu_van", r"Ngữ văn:\s*(\d+(?:\.\d+)?)"), + ("vat_ly", r"Vật lí:\s*(\d+(?:\.\d+)?)"), + ("hoa_hoc", r"Hóa học:\s*(\d+(?:\.\d+)?)"), + ("sinh_hoc", r"Sinh học:\s*(\d+(?:\.\d+)?)"), + ("khtn", r"KHTN:\s*(\d+(?:\.\d+)?)"), + ("lich_su", r"Lịch sử:\s*(\d+(?:\.\d+)?)"), + ("dia_ly", r"Địa lí:\s*(\d+(?:\.\d+)?)"), + ("gdcd", r"GDCD:\s*(\d+(?:\.\d+)?)"), + ("khxh", r"KHXH:\s*(\d+(?:\.\d+)?)"), + ("tieng_anh", r"Tiếng Anh:\s*(\d+(?:\.\d+)?)"), + ("tieng_phap", r"Tiếng Pháp:\s*(\d+(?:\.\d+)?)"), + ("tieng_nga", r"Tiếng Nga:\s*(\d+(?:\.\d+)?)"), + ("tieng_duc", r"Tiếng Đức:\s*(\d+(?:\.\d+)?)"), + ("tieng_nhat", r"Tiếng Nhật:\s*(\d+(?:\.\d+)?)"), + ("tieng_trung", r"Tiếng Trung:\s*(\d+(?:\.\d+)?)"), +]; + +// --------------------------------------------------------------------------- +// Tests +// --------------------------------------------------------------------------- + +#[cfg(test)] +mod tests { + use super::*; + + /// The INSERT placeholder count, the named column list and the field + /// constants must agree, or rows silently land in the wrong columns. + #[test] + fn insert_matches_field_order() { + assert_eq!(PARAM_COUNT, 22); + assert_eq!(INSERT_SQL.matches('?').count(), PARAM_COUNT); + + let named = INSERT_SQL + .split_once('(') + .and_then(|(_, rest)| rest.split_once(')')) + .map(|(cols, _)| cols) + .expect("INSERT must contain a column list"); + + let listed: Vec<&str> = named + .split(',') + .map(str::trim) + .filter(|s| !s.is_empty()) + .collect(); + + let expected: Vec<&str> = IDENTITY_FIELDS + .iter() + .chain(SCORE_FIELDS.iter()) + .copied() + .collect(); + + assert_eq!(listed, expected); + } + + /// Every subject column must have a pattern and vice versa. + #[test] + fn score_patterns_cover_score_fields() { + assert_eq!(SCORE_PATTERNS.len(), SCORE_FIELDS.len()); + for (field, _) in SCORE_PATTERNS { + assert!( + SCORE_FIELDS.contains(field), + "pattern {field} has no column" + ); + } + for field in SCORE_FIELDS { + assert!( + SCORE_PATTERNS.iter().any(|(f, _)| f == field), + "column {field} has no pattern" + ); + } + } + + /// Every column named in the DDL must be bound by the INSERT. + #[test] + fn ddl_columns_match_insert() { + for field in IDENTITY_FIELDS.iter().chain(SCORE_FIELDS.iter()) { + assert!(DDL.contains(field), "DDL missing column {field}"); + } + } + + #[test] + fn score_patterns_compile() { + for (field, src) in SCORE_PATTERNS { + regex::Regex::new(src).unwrap_or_else(|e| panic!("{field}: {e}")); + } + } +} diff --git a/2016/tools/xlsxread/src/transform.rs b/2016/tools/xlsxread/src/transform.rs index c688be4..1ea2d68 100644 --- a/2016/tools/xlsxread/src/transform.rs +++ b/2016/tools/xlsxread/src/transform.rs @@ -15,22 +15,25 @@ use crate::error::BuildError; // --------------------------------------------------------------------------- pub struct CompiledPatterns { - /// Ordered list so INSERT column order is deterministic + /// One compiled regex per subject column in `schema::SCORE_PATTERNS`. pub patterns: Vec<(String, Regex)>, } impl CompiledPatterns { - pub fn new(scores: &HashMap) -> Result { - let mut patterns = Vec::with_capacity(scores.len()); - for (field, src) in scores { + /// Compile the canonical subject patterns once at startup. + /// + /// All 16 patterns run against every dataset. A subject that did not exist + /// in a given exam year simply never matches and stays NULL — the parity + /// check asserts those counts are exactly zero rather than assuming it. + pub fn new() -> Result { + let mut patterns = Vec::with_capacity(crate::schema::SCORE_PATTERNS.len()); + for (field, src) in crate::schema::SCORE_PATTERNS { let re = Regex::new(src).map_err(|e| BuildError::Regex { - pattern: src.clone(), + pattern: (*src).to_string(), source: e, })?; - patterns.push((field.clone(), re)); + patterns.push(((*field).to_string(), re)); } - // Sort for deterministic order across HashMap iteration - patterns.sort_by(|a, b| a.0.cmp(&b.0)); Ok(Self { patterns }) } } @@ -314,11 +317,7 @@ mod tests { // --- Score parsing tests --- fn make_patterns() -> CompiledPatterns { - let mut map = HashMap::new(); - map.insert("toan".into(), r"Toán:\s*(\d+(?:\.\d+)?)".into()); - map.insert("ngu_van".into(), r"Ngữ văn:\s*(\d+(?:\.\d+)?)".into()); - map.insert("vat_ly".into(), r"Vật lí:\s*(\d+(?:\.\d+)?)".into()); - CompiledPatterns::new(&map).unwrap() + CompiledPatterns::new().unwrap() } #[test] diff --git a/2016/tools/xlsxread/src/writer.rs b/2016/tools/xlsxread/src/writer.rs index 07a83f0..e8fc17a 100644 --- a/2016/tools/xlsxread/src/writer.rs +++ b/2016/tools/xlsxread/src/writer.rs @@ -3,27 +3,24 @@ /// Mirrors build-lib.js createDb + the transaction loop in each build-database*.js. /// Stats output lines match the JS stdout exactly so existing CI log-greps still work. /// -/// thptqg2016 differences vs thptqg2017: -/// - INSERT includes ten_cum_thi and gioi_tinh after ngay_sinh -/// - Score columns are toan/ngu_van/vat_ly/hoa_hoc/sinh_hoc/lich_su/dia_ly/ -/// tieng_anh/tieng_phap/tieng_duc/tieng_nhat/tieng_trung -/// (no khtn, khxh, tieng_nga) +/// Every dataset writes the same canonical table (see `crate::schema`), so there +/// is exactly one insert path. Columns a dataset carries no data for bind NULL. use std::fs; use std::path::Path; use rusqlite::{params_from_iter, Connection, ToSql}; -use crate::config::DatasetConfig; use crate::error::BuildError; +use crate::schema; use crate::transform::ParsedRow; // --------------------------------------------------------------------------- // DB initialisation — mirrors build-lib.js createDb (delete + recreate) // --------------------------------------------------------------------------- -/// Open (or recreate) the output SQLite database, execute the DDL from config, +/// Open (or recreate) the output SQLite database, execute the canonical DDL, /// and return the open connection ready for inserts. -pub fn open_db(db_path: &Path, cfg: &DatasetConfig) -> Result { +pub fn open_db(db_path: &Path) -> Result { // Mirror Node behaviour: delete existing file before creating (build-lib.js:54) if db_path.exists() { fs::remove_file(db_path).map_err(|e| BuildError::Io { @@ -43,92 +40,22 @@ pub fn open_db(db_path: &Path, cfg: &DatasetConfig) -> Result Result<(), BuildError> { + let mut params: Vec> = Vec::with_capacity(schema::PARAM_COUNT); -/// thptqg2016 score columns (12 fields: tieng_duc/tieng_nhat present; no khtn/khxh/tieng_nga). -pub const SCORE_FIELDS_2016: &[&str] = &[ - "toan", - "ngu_van", - "vat_ly", - "hoa_hoc", - "sinh_hoc", - "lich_su", - "dia_ly", - "tieng_anh", - "tieng_phap", - "tieng_duc", - "tieng_nhat", - "tieng_trung", -]; - -/// Alias kept so thptqg2017 callers that import SCORE_FIELDS continue to compile. -pub const SCORE_FIELDS: &[&str] = SCORE_FIELDS_2017; - -// --------------------------------------------------------------------------- -// Insert a single parsed row (thptqg2017 schema — no ten_cum_thi / gioi_tinh) -// --------------------------------------------------------------------------- - -/// Bind all fields from `row` into the prepared statement and execute it. -/// Positional params: so_bao_danh, ho_ten, ho_ten_ascii, ngay_sinh, -pub fn insert_row( - conn: &Connection, - sql: &str, - row: &ParsedRow, - score_fields: &[&str], -) -> Result<(), BuildError> { - let mut params: Vec> = Vec::with_capacity(4 + score_fields.len()); - params.push(Box::new(row.so_bao_danh.clone())); - params.push(Box::new(row.ho_ten.clone())); - params.push(Box::new(row.ho_ten_ascii.clone())); - params.push(Box::new(row.ngay_sinh.clone())); - - for field in score_fields { - let val: Option = row.scores.get(*field).copied(); - params.push(Box::new(val)); - } - - conn.execute(sql, params_from_iter(params.iter().map(|p| p.as_ref())))?; - Ok(()) -} - -// --------------------------------------------------------------------------- -// Insert a single parsed row (thptqg2016 schema — includes ten_cum_thi + gioi_tinh) -// --------------------------------------------------------------------------- - -/// Bind all fields from `row` into the prepared statement for the thptqg2016 schema. -/// Positional params: so_bao_danh, ho_ten, ho_ten_ascii, ngay_sinh, ten_cum_thi, -/// gioi_tinh, -pub fn insert_row_2016( - conn: &Connection, - sql: &str, - row: &ParsedRow, - score_fields: &[&str], -) -> Result<(), BuildError> { - let mut params: Vec> = Vec::with_capacity(6 + score_fields.len()); params.push(Box::new(row.so_bao_danh.clone())); params.push(Box::new(row.ho_ten.clone())); params.push(Box::new(row.ho_ten_ascii.clone())); @@ -136,12 +63,17 @@ pub fn insert_row_2016( params.push(Box::new(row.ten_cum_thi.clone())); params.push(Box::new(row.gioi_tinh.clone())); - for field in score_fields { + for field in schema::SCORE_FIELDS { let val: Option = row.scores.get(*field).copied(); params.push(Box::new(val)); } - conn.execute(sql, params_from_iter(params.iter().map(|p| p.as_ref())))?; + debug_assert_eq!(params.len(), schema::PARAM_COUNT); + + conn.execute( + schema::INSERT_SQL, + params_from_iter(params.iter().map(|p| p.as_ref())), + )?; Ok(()) } diff --git a/2016/tools/xlsxread/tests/golden.rs b/2016/tools/xlsxread/tests/golden.rs index 0038945..47d1309 100644 --- a/2016/tools/xlsxread/tests/golden.rs +++ b/2016/tools/xlsxread/tests/golden.rs @@ -581,8 +581,7 @@ fn run_build_cmd(input_dir: &Path, db_path: &Path, config_name: &str) { let cfg = xlsxread::config::load_config(&cfg_path) .unwrap_or_else(|e| panic!("load config {config_name}: {e}")); - let patterns = - xlsxread::transform::CompiledPatterns::new(&cfg.scores).expect("compile patterns"); + let patterns = xlsxread::transform::CompiledPatterns::new().expect("compile patterns"); // Collect files let mut files: Vec = std::fs::read_dir(input_dir) @@ -602,7 +601,7 @@ fn run_build_cmd(input_dir: &Path, db_path: &Path, config_name: &str) { .collect(); files.sort(); - let conn = xlsxread::writer::open_db(db_path, &cfg).expect("open db"); + let conn = xlsxread::writer::open_db(db_path).expect("open db"); conn.execute_batch("BEGIN").unwrap(); for file in &files { @@ -632,12 +631,7 @@ fn run_build_cmd(input_dir: &Path, db_path: &Path, config_name: &str) { return; } let row = xlsxread::transform::transform_row(raw, &cfg, &patterns); - let _ = xlsxread::writer::insert_row( - &conn, - &cfg.insert.sql, - &row, - xlsxread::writer::SCORE_FIELDS, - ); + let _ = xlsxread::writer::insert_row(&conn, &row); }) .expect("process file"); } diff --git a/2017/scripts/db-stats.js b/2017/scripts/db-stats.js new file mode 100644 index 0000000..8582821 --- /dev/null +++ b/2017/scripts/db-stats.js @@ -0,0 +1,79 @@ +#!/usr/bin/env node +/** + * Dump per-dataset statistics from built SQLite databases as JSON. + * + * Used twice: once against the pre-refactor databases to capture a baseline, + * and again after the schema unification. Comparing the two outputs is what + * proves no score data was lost or invented. + * + * Schema-agnostic on purpose — columns come from PRAGMA table_info, so the same + * script runs against the old 18/20-column tables and the new 22-column one. + * + * Usage: + * node db-stats.js